新增 Pinterest 参考模式:独立于 Google Trends 的完整链路(12国种子词池 / LLM搜索词json_schema+防重复+已用词限100 / 并发爬图 / 多模态分析→原创简报 / 生图带爬取图参考图生图 / UI流程选择)
This commit is contained in:
@@ -18,6 +18,9 @@ logs/
|
|||||||
# 缓存(可重新采集/生成)
|
# 缓存(可重新采集/生成)
|
||||||
.cache/
|
.cache/
|
||||||
|
|
||||||
|
# Pinterest 登录态(首次手动登录后缓存,不入库)
|
||||||
|
pinterest_scraper/.chrome_session/
|
||||||
|
|
||||||
# 系统
|
# 系统
|
||||||
.DS_Store
|
.DS_Store
|
||||||
Thumbs.db
|
Thumbs.db
|
||||||
|
|||||||
+14
-8
@@ -33,15 +33,21 @@ seed_provider_cfg:
|
|||||||
trending_context_limit: 15
|
trending_context_limit: 15
|
||||||
history_limit: 20
|
history_limit: 20
|
||||||
|
|
||||||
# —— Pinterest(官方 API v5,可选;默认关闭)——
|
# —— Pinterest 参考模式(独立于 Google Trends 采集,UI「Pinterest 参考模式」入口)——
|
||||||
|
# 流程:各国独立种子词池(configs/pinterest/<CC>.yaml) → LLM 生成搜索词(json_schema+防重复)
|
||||||
|
# → 过滤 → scraper 爬取图片 → LLM 分析图片 → 构造提示词 → 生成设计
|
||||||
pinterest:
|
pinterest:
|
||||||
enabled: false
|
enabled: true
|
||||||
access_token: "" # 填 Bearer Token,或用环境变量 PINTEREST_ACCESS_TOKEN
|
provider: openai # 搜索词/图片分析用 LLM 提供商(openai / mock)
|
||||||
search_keywords:
|
search_terms_per_run: 10 # 每次运行 LLM 生成的搜索词数量
|
||||||
- "trending fashion"
|
seed_sample: 40 # 每次从国家种子池随机抽取多少个种子词给 LLM
|
||||||
- "streetwear"
|
max_used_terms_in_prompt: 100 # 已用搜索词最多注入 LLM 提示词的个数(防 token 超限)
|
||||||
- "cottagecore"
|
images_per_term: 40 # 每个搜索词爬取图片数量
|
||||||
page_size: 25
|
analyze_per_term: 6 # 每个搜索词最多分析几张图(生成设计简报)
|
||||||
|
max_designs: 10 # 本次最多生成多少个设计
|
||||||
|
ref_images_per_design: 1 # 生图时每个设计附带几张爬取图作为参考(发给生图模型)
|
||||||
|
scrape_concurrency: 2 # 同时爬取几个搜索词(每个会开一个 Chrome 窗口)
|
||||||
|
headless: false # 爬取时是否无头(false=显示 Chrome 窗口,首次需手动登录)
|
||||||
|
|
||||||
# 跨源融合权重(按 source 标签,无需和为 1)
|
# 跨源融合权重(按 source 标签,无需和为 1)
|
||||||
# 已下调 gt_trending(泛国家热点只作微弱信号),主力偏向 style+related(可印花型词)。
|
# 已下调 gt_trending(泛国家热点只作微弱信号),主力偏向 style+related(可印花型词)。
|
||||||
|
|||||||
@@ -0,0 +1,43 @@
|
|||||||
|
# Pinterest 参考模式种子词(面向视觉灵感,非热点关键词)
|
||||||
|
# 约束:避开品牌/角色/名人/宗教/国旗/酒精等侵权与敏感项
|
||||||
|
seeds:
|
||||||
|
- australian wildlife
|
||||||
|
- surf culture
|
||||||
|
- outback landscape
|
||||||
|
- great barrier reef
|
||||||
|
- sydney harbour
|
||||||
|
- koala
|
||||||
|
- kangaroo
|
||||||
|
- beach lifestyle
|
||||||
|
- tropical rainforest
|
||||||
|
- coastal australia
|
||||||
|
- aussie retro
|
||||||
|
- australian birds
|
||||||
|
- uluru sunset
|
||||||
|
- australian bush
|
||||||
|
- sydney opera house
|
||||||
|
- australian beach
|
||||||
|
- coral reef
|
||||||
|
- australian flora
|
||||||
|
- eucalyptus
|
||||||
|
- australian summer
|
||||||
|
- surfboard retro
|
||||||
|
- australian outback road
|
||||||
|
- kangaroo silhouette
|
||||||
|
- australian coast
|
||||||
|
- bondi beach
|
||||||
|
- australian desert
|
||||||
|
- native australian plants
|
||||||
|
- australian wildlife art
|
||||||
|
- beach sunset
|
||||||
|
- australian retro poster
|
||||||
|
- great ocean road
|
||||||
|
- australian mountains
|
||||||
|
- tropical fish
|
||||||
|
- australian birds art
|
||||||
|
- surf retro
|
||||||
|
- australian landscape
|
||||||
|
- coastal walk
|
||||||
|
- australian animals
|
||||||
|
- beach house retro
|
||||||
|
- australian minimal
|
||||||
@@ -0,0 +1,43 @@
|
|||||||
|
# Pinterest 参考模式种子词(面向视觉灵感,非热点关键词)
|
||||||
|
# 约束:避开品牌/角色/名人/宗教/国旗/酒精等侵权与敏感项
|
||||||
|
seeds:
|
||||||
|
- brazilian tropical
|
||||||
|
- amazon rainforest
|
||||||
|
- brazilian street art
|
||||||
|
- carnival colors
|
||||||
|
- tropical birds
|
||||||
|
- brazilian flora
|
||||||
|
- favela art
|
||||||
|
- samba culture
|
||||||
|
- brazilian beach
|
||||||
|
- jaguar
|
||||||
|
- toucan
|
||||||
|
- brazilian retro
|
||||||
|
- tropical leaves
|
||||||
|
- brazilian wildlife
|
||||||
|
- brazilian coast
|
||||||
|
- rio landscape
|
||||||
|
- brazilian patterns
|
||||||
|
- tropical sunset
|
||||||
|
- brazilian birds
|
||||||
|
- brazilian art
|
||||||
|
- amazon wildlife
|
||||||
|
- brazilian minimal
|
||||||
|
- tropical flowers
|
||||||
|
- brazilian street style
|
||||||
|
- brazilian retro poster
|
||||||
|
- brazilian nature
|
||||||
|
- brazilian beach sunset
|
||||||
|
- brazilian architecture
|
||||||
|
- tropical fish
|
||||||
|
- brazilian forest
|
||||||
|
- brazilian folk art
|
||||||
|
- brazilian textiles
|
||||||
|
- brazilian mountains
|
||||||
|
- brazilian wildlife art
|
||||||
|
- tropical retro
|
||||||
|
- brazilian coast retro
|
||||||
|
- brazilian floral
|
||||||
|
- brazilian landscape
|
||||||
|
- brazilian summer
|
||||||
|
- brazilian retro design
|
||||||
@@ -0,0 +1,43 @@
|
|||||||
|
# Pinterest 参考模式种子词(面向视觉灵感,非热点关键词)
|
||||||
|
# 约束:避开品牌/角色/名人/宗教/国旗/酒精等侵权与敏感项
|
||||||
|
seeds:
|
||||||
|
- canadian wilderness
|
||||||
|
- northern lights
|
||||||
|
- canadian wildlife
|
||||||
|
- moose
|
||||||
|
- polar bear
|
||||||
|
- mountain lakes
|
||||||
|
- cottage country
|
||||||
|
- canadian retro
|
||||||
|
- hockey culture
|
||||||
|
- coastal canada
|
||||||
|
- canadian maple
|
||||||
|
- rocky mountains
|
||||||
|
- canadian forest
|
||||||
|
- canadian birds
|
||||||
|
- loon
|
||||||
|
- canadian canoe
|
||||||
|
- banff landscape
|
||||||
|
- canadian winter
|
||||||
|
- snowboarding retro
|
||||||
|
- canadian fishing
|
||||||
|
- maple forest
|
||||||
|
- canadian coast
|
||||||
|
- canadian summer
|
||||||
|
- canadian wildlife art
|
||||||
|
- niagara falls
|
||||||
|
- canadian prairie
|
||||||
|
- canadian retro poster
|
||||||
|
- canadian mountains
|
||||||
|
- canadian lake
|
||||||
|
- canadian minimal
|
||||||
|
- canadian autumn
|
||||||
|
- canadian wildlife illustration
|
||||||
|
- canadian cabin
|
||||||
|
- canadian trail
|
||||||
|
- canadian beach
|
||||||
|
- canadian skyline
|
||||||
|
- canadian retro travel
|
||||||
|
- canadian nature
|
||||||
|
- canadian wildlife retro
|
||||||
|
- canadian outdoor
|
||||||
@@ -0,0 +1,43 @@
|
|||||||
|
# Pinterest 参考模式种子词(面向视觉灵感,非热点关键词)
|
||||||
|
# 约束:避开品牌/角色/名人/宗教/国旗/酒精等侵权与敏感项
|
||||||
|
seeds:
|
||||||
|
- german folk art
|
||||||
|
- bavarian alpine
|
||||||
|
- black forest
|
||||||
|
- berlin street art
|
||||||
|
- cuckoo clock
|
||||||
|
- lederhosen
|
||||||
|
- german castle
|
||||||
|
- autobahn retro
|
||||||
|
- nordic minimalism
|
||||||
|
- german typography
|
||||||
|
- berlin wall art
|
||||||
|
- german mountains
|
||||||
|
- bavarian patterns
|
||||||
|
- german wildlife
|
||||||
|
- german retro poster
|
||||||
|
- german forest
|
||||||
|
- german architecture
|
||||||
|
- german countryside
|
||||||
|
- german birds
|
||||||
|
- german retro travel
|
||||||
|
- german minimal
|
||||||
|
- german coast
|
||||||
|
- german lakes
|
||||||
|
- german folk patterns
|
||||||
|
- german street style
|
||||||
|
- german nature
|
||||||
|
- german retro car
|
||||||
|
- german castle silhouette
|
||||||
|
- german winter
|
||||||
|
- german autumn
|
||||||
|
- german summer
|
||||||
|
- german floral
|
||||||
|
- german landscape
|
||||||
|
- german retro design
|
||||||
|
- german wildlife art
|
||||||
|
- german mountains retro
|
||||||
|
- german folk embroidery
|
||||||
|
- german minimal design
|
||||||
|
- german coastal
|
||||||
|
- german retro typography
|
||||||
@@ -0,0 +1,43 @@
|
|||||||
|
# Pinterest 参考模式种子词(面向视觉灵感,非热点关键词)
|
||||||
|
# 约束:避开品牌/角色/名人/宗教/国旗/酒精等侵权与敏感项
|
||||||
|
seeds:
|
||||||
|
- spanish tiles
|
||||||
|
- flamenco
|
||||||
|
- andalusian architecture
|
||||||
|
- spanish retro
|
||||||
|
- mediterranean coast
|
||||||
|
- spanish ceramics
|
||||||
|
- paella
|
||||||
|
- spanish guitar
|
||||||
|
- olive groves
|
||||||
|
- spanish countryside
|
||||||
|
- spanish floral
|
||||||
|
- spanish retro poster
|
||||||
|
- spanish wildlife
|
||||||
|
- spanish mountains
|
||||||
|
- spanish coast
|
||||||
|
- spanish minimal
|
||||||
|
- spanish architecture
|
||||||
|
- spanish street style
|
||||||
|
- spanish nature
|
||||||
|
- spanish retro design
|
||||||
|
- spanish birds
|
||||||
|
- spanish summer
|
||||||
|
- spanish landscape
|
||||||
|
- spanish folk art
|
||||||
|
- spanish pottery
|
||||||
|
- spanish retro travel
|
||||||
|
- spanish garden
|
||||||
|
- spanish tiles patterns
|
||||||
|
- spanish coastal
|
||||||
|
- spanish wildlife art
|
||||||
|
- spanish traditional patterns
|
||||||
|
- spanish retro typography
|
||||||
|
- spanish flowers
|
||||||
|
- spanish beach
|
||||||
|
- spanish countryside retro
|
||||||
|
- spanish folk patterns
|
||||||
|
- spanish minimal design
|
||||||
|
- spanish retro poster travel
|
||||||
|
- spanish mountains retro
|
||||||
|
- spanish art
|
||||||
@@ -0,0 +1,43 @@
|
|||||||
|
# Pinterest 参考模式种子词(面向视觉灵感,非热点关键词)
|
||||||
|
# 约束:避开品牌/角色/名人/宗教/国旗/酒精等侵权与敏感项
|
||||||
|
seeds:
|
||||||
|
- london street style
|
||||||
|
- british punk
|
||||||
|
- victorian botanical
|
||||||
|
- english countryside
|
||||||
|
- london skyline
|
||||||
|
- mod fashion
|
||||||
|
- britpop aesthetic
|
||||||
|
- royal guard
|
||||||
|
- tea culture
|
||||||
|
- coastal britain
|
||||||
|
- rock music retro
|
||||||
|
- london underground
|
||||||
|
- british seaside
|
||||||
|
- punk rock
|
||||||
|
- union jack vintage
|
||||||
|
- british wildlife
|
||||||
|
- scottish highlands
|
||||||
|
- welsh coast
|
||||||
|
- british retro poster
|
||||||
|
- london fashion week street
|
||||||
|
- english garden
|
||||||
|
- british pub sign
|
||||||
|
- london bridge
|
||||||
|
- british weather
|
||||||
|
- vintage british travel
|
||||||
|
- oxford academia
|
||||||
|
- british rock band retro
|
||||||
|
- london graffiti
|
||||||
|
- british countryside walk
|
||||||
|
- coastal cliffs
|
||||||
|
- british birds
|
||||||
|
- english heritage
|
||||||
|
- british floral
|
||||||
|
- london night
|
||||||
|
- british football retro
|
||||||
|
- cricket vintage
|
||||||
|
- british seaside pier
|
||||||
|
- london architecture
|
||||||
|
- english tea party
|
||||||
|
- british minimal
|
||||||
@@ -0,0 +1,43 @@
|
|||||||
|
# Pinterest 参考模式种子词(面向视觉灵感,非热点关键词)
|
||||||
|
# 约束:避开品牌/角色/名人/宗教/国旗/酒精等侵权与敏感项
|
||||||
|
seeds:
|
||||||
|
- italian renaissance
|
||||||
|
- tuscan landscape
|
||||||
|
- venetian mask
|
||||||
|
- italian retro
|
||||||
|
- mediterranean style
|
||||||
|
- roman architecture
|
||||||
|
- italian ceramics
|
||||||
|
- amalfi coast
|
||||||
|
- pasta
|
||||||
|
- italian espresso
|
||||||
|
- dolomites
|
||||||
|
- sicilian patterns
|
||||||
|
- italian countryside
|
||||||
|
- italian floral
|
||||||
|
- italian retro poster
|
||||||
|
- italian wildlife
|
||||||
|
- italian coast
|
||||||
|
- italian minimal
|
||||||
|
- italian architecture
|
||||||
|
- italian street style
|
||||||
|
- italian nature
|
||||||
|
- italian retro design
|
||||||
|
- italian birds
|
||||||
|
- italian summer
|
||||||
|
- italian landscape
|
||||||
|
- italian folk art
|
||||||
|
- italian pottery
|
||||||
|
- italian retro travel
|
||||||
|
- italian garden
|
||||||
|
- italian tiles patterns
|
||||||
|
- italian coastal
|
||||||
|
- italian wildlife art
|
||||||
|
- italian traditional patterns
|
||||||
|
- italian retro typography
|
||||||
|
- italian flowers
|
||||||
|
- italian beach
|
||||||
|
- italian countryside retro
|
||||||
|
- italian folk patterns
|
||||||
|
- italian minimal design
|
||||||
|
- italian art
|
||||||
@@ -0,0 +1,43 @@
|
|||||||
|
# Pinterest 参考模式种子词(面向视觉灵感,非热点关键词)
|
||||||
|
# 约束:避开品牌/角色/名人/宗教/国旗/酒精等侵权与敏感项
|
||||||
|
seeds:
|
||||||
|
- 和柄モチーフ
|
||||||
|
- 浮世絵デザイン
|
||||||
|
- レトロポップ
|
||||||
|
- ミニマルラインアート
|
||||||
|
- 花柄イラスト
|
||||||
|
- 猫イラスト
|
||||||
|
- 富士山グラフィック
|
||||||
|
- 渋谷ストリートスタイル
|
||||||
|
- 京都和風
|
||||||
|
- 桜モチーフ
|
||||||
|
- 波紋デザイン
|
||||||
|
- 神社鳥居
|
||||||
|
- 星座イラスト
|
||||||
|
- かわいい動物
|
||||||
|
- 昭和レトロ
|
||||||
|
- 大正ロマン
|
||||||
|
- 千鳥格子
|
||||||
|
- 金魚
|
||||||
|
- 提灯
|
||||||
|
- 和菓子
|
||||||
|
- 招き猫
|
||||||
|
- だるま
|
||||||
|
- 風鈴
|
||||||
|
- 浴衣柄
|
||||||
|
- 歌舞伎モチーフ
|
||||||
|
- 水墨画
|
||||||
|
- 折り紙
|
||||||
|
- 提灯祭り
|
||||||
|
- 紅葉
|
||||||
|
- 竹
|
||||||
|
- 鶴
|
||||||
|
- 鯉のぼり
|
||||||
|
- 梅
|
||||||
|
- 雪景色
|
||||||
|
- 夏祭り
|
||||||
|
- 縁日
|
||||||
|
- 雷門
|
||||||
|
- 五重塔
|
||||||
|
- 和太鼓
|
||||||
|
- 風神雷神
|
||||||
@@ -0,0 +1,43 @@
|
|||||||
|
# Pinterest 参考模式种子词(面向视觉灵感,非热点关键词)
|
||||||
|
# 约束:避开品牌/角色/名人/宗教/国旗/酒精等侵权与敏感项
|
||||||
|
seeds:
|
||||||
|
- mexican folk art
|
||||||
|
- talavera patterns
|
||||||
|
- cacti desert
|
||||||
|
- aztec patterns
|
||||||
|
- mexican food
|
||||||
|
- sombrero
|
||||||
|
- papel picado
|
||||||
|
- otomi patterns
|
||||||
|
- mexican retro
|
||||||
|
- colorful mexican tiles
|
||||||
|
- mexican embroidery
|
||||||
|
- marigold flowers
|
||||||
|
- mexican birds
|
||||||
|
- desert sunset
|
||||||
|
- mexican pottery
|
||||||
|
- lucha libre retro
|
||||||
|
- mexican architecture
|
||||||
|
- tropical mexico
|
||||||
|
- mexican beach
|
||||||
|
- mayan patterns
|
||||||
|
- mexican textiles
|
||||||
|
- cactus illustration
|
||||||
|
- mexican skull art
|
||||||
|
- fiesta colors
|
||||||
|
- mexican landscape
|
||||||
|
- mexican flowers
|
||||||
|
- serape blanket
|
||||||
|
- mexican market
|
||||||
|
- colonial mexico
|
||||||
|
- mexican retro poster
|
||||||
|
- agave plant
|
||||||
|
- mexican street art
|
||||||
|
- chiapas textiles
|
||||||
|
- mexican wildlife
|
||||||
|
- oaxaca patterns
|
||||||
|
- mexican sunset
|
||||||
|
- mexican folk patterns
|
||||||
|
- tropical birds
|
||||||
|
- mexican handcraft
|
||||||
|
- mexican minimal
|
||||||
@@ -0,0 +1,43 @@
|
|||||||
|
# Pinterest 参考模式种子词(面向视觉灵感,非热点关键词)
|
||||||
|
# 约束:避开品牌/角色/名人/宗教/国旗/酒精等侵权与敏感项
|
||||||
|
seeds:
|
||||||
|
- polish folk art
|
||||||
|
- polish pottery
|
||||||
|
- bialowieza forest
|
||||||
|
- polish mountains
|
||||||
|
- wawel castle
|
||||||
|
- slavic patterns
|
||||||
|
- amber
|
||||||
|
- polish retro
|
||||||
|
- vistula river
|
||||||
|
- polish embroidery
|
||||||
|
- polish folk patterns
|
||||||
|
- polish wildlife
|
||||||
|
- polish countryside
|
||||||
|
- polish architecture
|
||||||
|
- polish retro poster
|
||||||
|
- polish birds
|
||||||
|
- polish nature
|
||||||
|
- polish coast
|
||||||
|
- polish lakes
|
||||||
|
- polish minimal
|
||||||
|
- slavic embroidery
|
||||||
|
- polish folk flowers
|
||||||
|
- polish retro design
|
||||||
|
- polish winter
|
||||||
|
- polish autumn
|
||||||
|
- polish summer
|
||||||
|
- polish landscape
|
||||||
|
- polish folk art patterns
|
||||||
|
- polish mountains retro
|
||||||
|
- polish wildlife art
|
||||||
|
- polish folk costume
|
||||||
|
- polish retro travel
|
||||||
|
- polish forest
|
||||||
|
- polish coastal
|
||||||
|
- polish folk pottery
|
||||||
|
- polish minimal design
|
||||||
|
- polish retro typography
|
||||||
|
- polish folk birds
|
||||||
|
- polish countryside retro
|
||||||
|
- polish traditional patterns
|
||||||
@@ -0,0 +1,43 @@
|
|||||||
|
# Pinterest 参考模式种子词(面向视觉灵感,非热点关键词)
|
||||||
|
# 约束:避开品牌/角色/名人/宗教/国旗/酒精等侵权与敏感项
|
||||||
|
seeds:
|
||||||
|
- arabic calligraphy
|
||||||
|
- desert dunes
|
||||||
|
- saudi architecture
|
||||||
|
- middle eastern patterns
|
||||||
|
- camel
|
||||||
|
- oasis
|
||||||
|
- arabic geometric patterns
|
||||||
|
- traditional arabic design
|
||||||
|
- desert night sky
|
||||||
|
- palm oasis
|
||||||
|
- arabic floral patterns
|
||||||
|
- desert sunset
|
||||||
|
- saudi retro
|
||||||
|
- arabian horses
|
||||||
|
- desert wildlife
|
||||||
|
- arabic tiles
|
||||||
|
- saudi coast
|
||||||
|
- red sea
|
||||||
|
- arabic lantern
|
||||||
|
- desert retro poster
|
||||||
|
- saudi minimal
|
||||||
|
- arabic typography
|
||||||
|
- desert landscape
|
||||||
|
- saudi mountains
|
||||||
|
- arabic birds
|
||||||
|
- desert stars
|
||||||
|
- saudi nature
|
||||||
|
- arabic architecture
|
||||||
|
- desert flowers
|
||||||
|
- saudi wildlife
|
||||||
|
- arabic patterns retro
|
||||||
|
- saudi beach
|
||||||
|
- desert retro
|
||||||
|
- arabic minimal design
|
||||||
|
- saudi folk art
|
||||||
|
- desert caravan
|
||||||
|
- arabic pottery
|
||||||
|
- saudi retro poster
|
||||||
|
- desert oasis illustration
|
||||||
|
- arabic geometric art
|
||||||
@@ -0,0 +1,44 @@
|
|||||||
|
# Pinterest 参考模式种子词(面向视觉灵感,非热点关键词)
|
||||||
|
# 用途:LLM 据此生成 Pinterest 搜索词 → 爬取图片 → 分析 → 生成 T 恤设计
|
||||||
|
# 约束:避开品牌/角色/名人/宗教/国旗/酒精等侵权与敏感项
|
||||||
|
seeds:
|
||||||
|
- vintage 70s retro
|
||||||
|
- desert southwest
|
||||||
|
- coastal beach vibes
|
||||||
|
- botanical illustration
|
||||||
|
- retro surf culture
|
||||||
|
- mountain landscape
|
||||||
|
- western cowboy
|
||||||
|
- minimalist line art
|
||||||
|
- american diner retro
|
||||||
|
- national park
|
||||||
|
- road trip
|
||||||
|
- skate culture
|
||||||
|
- floral watercolor
|
||||||
|
- celestial night sky
|
||||||
|
- mid-century modern
|
||||||
|
- boho festival
|
||||||
|
- grunge aesthetic
|
||||||
|
- cottagecore
|
||||||
|
- y2k fashion
|
||||||
|
- streetwear graphic
|
||||||
|
- retro arcade
|
||||||
|
- vintage travel poster
|
||||||
|
- abstract geometric
|
||||||
|
- hand drawn doodle
|
||||||
|
- retro sunset
|
||||||
|
- palm tree summer
|
||||||
|
- wild west
|
||||||
|
- space exploration
|
||||||
|
- ocean waves
|
||||||
|
- forest wildlife
|
||||||
|
- retro typography
|
||||||
|
- pop art
|
||||||
|
- art deco
|
||||||
|
- psychedelic
|
||||||
|
- vintage motorcycle
|
||||||
|
- retro camper van
|
||||||
|
- american classic car
|
||||||
|
- baseball retro
|
||||||
|
- basketball street
|
||||||
|
- hiking adventure
|
||||||
@@ -58,6 +58,39 @@ def build_graph():
|
|||||||
return builder.compile()
|
return builder.compile()
|
||||||
|
|
||||||
|
|
||||||
|
def build_pinterest_graph():
|
||||||
|
"""Pinterest 参考模式图(独立于 Google Trends 采集链路):
|
||||||
|
pinterest_search → pinterest_scrape → pinterest_analyze → compose → product
|
||||||
|
→ oss_upload → seed_shot → template_export
|
||||||
|
"""
|
||||||
|
from graph.nodes import (
|
||||||
|
pinterest_analyze_node,
|
||||||
|
pinterest_scrape_node,
|
||||||
|
pinterest_search_node,
|
||||||
|
)
|
||||||
|
|
||||||
|
builder = StateGraph(AgentState)
|
||||||
|
builder.add_node("pinterest_search", pinterest_search_node)
|
||||||
|
builder.add_node("pinterest_scrape", pinterest_scrape_node)
|
||||||
|
builder.add_node("pinterest_analyze", pinterest_analyze_node)
|
||||||
|
builder.add_node("compose", compose_node)
|
||||||
|
builder.add_node("product", product_node)
|
||||||
|
builder.add_node("oss_upload", oss_upload_node)
|
||||||
|
builder.add_node("seed_shot", seed_shot_node)
|
||||||
|
builder.add_node("template_export", template_export_node)
|
||||||
|
|
||||||
|
builder.add_edge("__start__", "pinterest_search")
|
||||||
|
builder.add_edge("pinterest_search", "pinterest_scrape")
|
||||||
|
builder.add_edge("pinterest_scrape", "pinterest_analyze")
|
||||||
|
builder.add_edge("pinterest_analyze", "compose")
|
||||||
|
builder.add_edge("compose", "product")
|
||||||
|
builder.add_edge("product", "oss_upload")
|
||||||
|
builder.add_edge("oss_upload", "seed_shot")
|
||||||
|
builder.add_edge("seed_shot", "template_export")
|
||||||
|
builder.add_edge("template_export", END)
|
||||||
|
return builder.compile()
|
||||||
|
|
||||||
|
|
||||||
def run_country(
|
def run_country(
|
||||||
country: str,
|
country: str,
|
||||||
global_config: Dict[str, Any],
|
global_config: Dict[str, Any],
|
||||||
@@ -109,3 +142,46 @@ def run_country(
|
|||||||
|
|
||||||
result = compiled.invoke(state)
|
result = compiled.invoke(state)
|
||||||
return result
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
def run_pinterest_ref(
|
||||||
|
country: str,
|
||||||
|
global_config: Dict[str, Any],
|
||||||
|
project_root: Path,
|
||||||
|
output_root: Optional[Path] = None,
|
||||||
|
task_timestamp: Optional[str] = None,
|
||||||
|
) -> Dict[str, Any]:
|
||||||
|
"""Pinterest 参考模式入口:独立于 Google Trends 的完整流程。
|
||||||
|
|
||||||
|
种子词 → LLM 搜索词(json_schema + 动态注入防重复)→ 爬图 → LLM 分析图片
|
||||||
|
→ 设计简报 → 设计稿 → 产品图 → 上传 → 种草图 → 模板导出。
|
||||||
|
参数语义与 run_country 一致(project_root=数据根,output_root=产物根)。
|
||||||
|
"""
|
||||||
|
compiled = build_pinterest_graph()
|
||||||
|
cc = build_country_config(global_config, country, project_root)
|
||||||
|
prompts_dir = project_root / "prompts" / country
|
||||||
|
cache_dir = (output_root or project_root) / "output" / country
|
||||||
|
ts = task_timestamp or time.strftime("%Y%m%d_%H%M%S")
|
||||||
|
_base = ts
|
||||||
|
_i = 1
|
||||||
|
while (cache_dir / ts).exists(): # 时间戳文件夹唯一(同秒多任务防冲突/覆盖)
|
||||||
|
ts = f"{_base}_{_i}"
|
||||||
|
_i += 1
|
||||||
|
output_dir = cache_dir / ts
|
||||||
|
|
||||||
|
state: Dict[str, Any] = {
|
||||||
|
"country": country,
|
||||||
|
"config": global_config,
|
||||||
|
"country_config": cc,
|
||||||
|
"prompts_dir": str(prompts_dir),
|
||||||
|
"cache_dir": str(cache_dir),
|
||||||
|
"output_dir": str(output_dir),
|
||||||
|
"briefs": [],
|
||||||
|
"composite": [],
|
||||||
|
"designs": [],
|
||||||
|
"errors": [],
|
||||||
|
"stats": {},
|
||||||
|
"task_timestamp": ts,
|
||||||
|
"oss_seq": 0,
|
||||||
|
}
|
||||||
|
return compiled.invoke(state)
|
||||||
|
|||||||
@@ -145,3 +145,46 @@ class MockBackend:
|
|||||||
"style_seeds": _dedup_limit(style, max_style),
|
"style_seeds": _dedup_limit(style, max_style),
|
||||||
"related_seeds": _dedup_limit(related, max_related),
|
"related_seeds": _dedup_limit(related, max_related),
|
||||||
}
|
}
|
||||||
|
|
||||||
|
def generate_pinterest_terms(self, context: Dict[str, Any]) -> Dict[str, Any]:
|
||||||
|
"""规则生成 Pinterest 搜索词(零 API 成本):从种子词池随机取 + 两两组合增加多样性。"""
|
||||||
|
import random
|
||||||
|
seeds = [str(s).strip() for s in (context.get("seeds") or []) if str(s).strip()]
|
||||||
|
used = {str(u).strip().lower() for u in (context.get("used_terms") or [])}
|
||||||
|
count = int(context.get("count", 10))
|
||||||
|
pool = [s for s in seeds if s.lower() not in used]
|
||||||
|
random.shuffle(pool)
|
||||||
|
terms = pool[:count]
|
||||||
|
# 不足时用「种子词 + 风格词」组合补足(视觉导向,避免与已用重复)
|
||||||
|
style_tail = ["aesthetic", "style", "inspiration", "design", "vibe", "art"]
|
||||||
|
i = 0
|
||||||
|
while len(terms) < count and pool:
|
||||||
|
combo = f"{pool[i % len(pool)]} {style_tail[(i // len(pool)) % len(style_tail)]}"
|
||||||
|
if combo.lower() not in used and combo not in terms:
|
||||||
|
terms.append(combo)
|
||||||
|
i += 1
|
||||||
|
return {"search_terms": terms}
|
||||||
|
|
||||||
|
def analyze_pinterest_images(self, image_paths, term="", country=""):
|
||||||
|
"""规则生成设计简报(零 API 成本):按搜索词启发式推导风格/配色/构图。"""
|
||||||
|
from ..classify import classify, prompt_suggestion
|
||||||
|
cat = classify(term)
|
||||||
|
art_style, palette = derive_style_palette(term, country, category=cat)
|
||||||
|
motif = prompt_suggestion(term, cat).split(" --no ")[0].split(",")[0].strip()
|
||||||
|
composition = derive_composition(term, cat)
|
||||||
|
negative = ("no real people, no likeness of any person, no copyrighted characters, "
|
||||||
|
"no brand logos, no trademarks, no celebrity, no readable text unless safe")
|
||||||
|
n = max(1, len(image_paths or []))
|
||||||
|
paths = list(image_paths or [])
|
||||||
|
return [{
|
||||||
|
"topic": term,
|
||||||
|
"concept": f"(启发式兜底)围绕「{term}」做原创{art_style}风格印花",
|
||||||
|
"motif": motif,
|
||||||
|
"art_style": art_style,
|
||||||
|
"color_palette": palette,
|
||||||
|
"composition": composition,
|
||||||
|
"negative_prompt": negative,
|
||||||
|
# 生图参考:每条简报对应其来源爬取图(mock 按图逐张产出简报,顺序一一对应)
|
||||||
|
"ref_images": [str(paths[i])] if i < len(paths) else [],
|
||||||
|
"source": "pinterest",
|
||||||
|
} for i in range(n)]
|
||||||
|
|||||||
@@ -233,6 +233,118 @@ def build_user_prompt(country, topics, aesthetic_hint):
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
# —— Pinterest 参考模式:搜索词生成(json_schema 结构化 + 动态注入已用词防重复)——
|
||||||
|
PINTEREST_TERM_SYSTEM_PROMPT = """You are a Pinterest search-term generator for print-on-demand (POD) T-shirt design.
|
||||||
|
You turn seed words into diverse, visual, Pinterest-friendly search terms that will be used to scrape inspiration images.
|
||||||
|
|
||||||
|
RULES:
|
||||||
|
- Generate EXACTLY the requested number of search terms.
|
||||||
|
- Terms must be VISUAL / AESTHETIC concepts (style, motif, scene, color) suitable as T-shirt print inspiration.
|
||||||
|
- Terms must be DIVERSE and NON-OVERLAPPING: never repeat a concept, never give near-synonyms of each other.
|
||||||
|
- DO NOT repeat or closely paraphrase ANY of the "already used terms" provided in the user message.
|
||||||
|
- Use the country's local language where natural (e.g. Japanese for JP, Spanish for ES/MX), else English.
|
||||||
|
- Each term is 2-4 words, concise, no punctuation.
|
||||||
|
- COPYRIGHT-SAFE: no brands, no logos, no characters, no celebrities, no real persons, no franchises.
|
||||||
|
- AVOID: politics, religion, hate, violence, sexual content, alcohol, national flags.
|
||||||
|
|
||||||
|
Return JSON with the field "search_terms" (array of strings)."""
|
||||||
|
|
||||||
|
PINTEREST_TERM_SCHEMA = {
|
||||||
|
"name": "pinterest_search_terms",
|
||||||
|
"schema": {
|
||||||
|
"type": "object",
|
||||||
|
"properties": {
|
||||||
|
"search_terms": {
|
||||||
|
"type": "array",
|
||||||
|
"items": {"type": "string"},
|
||||||
|
"description": "Diverse, non-overlapping Pinterest search terms for T-shirt design inspiration",
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"required": ["search_terms"],
|
||||||
|
"additionalProperties": False,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def build_pinterest_term_user_prompt(context: Dict[str, Any]) -> str:
|
||||||
|
"""动态注入:种子词(灵感)+ 已用搜索词(禁止重复)+ 数量要求。"""
|
||||||
|
seeds = context.get("seeds", []) or []
|
||||||
|
used = context.get("used_terms", []) or []
|
||||||
|
count = int(context.get("count", 10))
|
||||||
|
lines = [
|
||||||
|
f"Country: {context.get('country', '')}",
|
||||||
|
f"Seed words (inspiration, may combine or extend): {', '.join(seeds)}",
|
||||||
|
"",
|
||||||
|
f"Already used terms — DO NOT repeat or paraphrase ANY of these: "
|
||||||
|
f"{', '.join(used) if used else '(none yet)'}",
|
||||||
|
"",
|
||||||
|
f"Generate {count} new, diverse, non-overlapping Pinterest search terms.",
|
||||||
|
]
|
||||||
|
return "\n".join(lines)
|
||||||
|
|
||||||
|
|
||||||
|
# —— Pinterest 参考模式:图片分析 → 原创设计简报(多模态)——
|
||||||
|
PINTEREST_ANALYZE_SYSTEM_PROMPT = """You are a POD (print-on-demand) T-shirt design analyst.
|
||||||
|
You receive Pinterest reference images for one search term. For each image, extract the VISUAL CONCEPT
|
||||||
|
(style, mood, motif, color palette, composition) that makes it appealing, then produce an ORIGINAL
|
||||||
|
T-shirt print design brief that captures that VIBE WITHOUT copying the image.
|
||||||
|
|
||||||
|
RULES:
|
||||||
|
- NEVER copy the image, never reproduce the exact artwork, characters, logos, or any text from it.
|
||||||
|
- Extract only the abstract style/mood/motif concept as inspiration.
|
||||||
|
- Produce an original, flat, print-ready design brief (no garment, no model, no background scene).
|
||||||
|
- COPYRIGHT-SAFE: no brands, no logos, no characters, no celebrities, no real persons, no franchises.
|
||||||
|
- AVOID: politics, religion, hate, violence, sexual content, alcohol, national flags.
|
||||||
|
- motif: English, concrete central subject of the print (e.g. "a smiling cat with a fish", "geometric mountain layers").
|
||||||
|
- art_style: English visual technique (e.g. "clean flat vector", "retro screen print").
|
||||||
|
- color_palette: English colors (e.g. "sunset orange, cream, dusty blue").
|
||||||
|
- composition: English layout (e.g. "centered emblem with balanced negative space").
|
||||||
|
- concept: Chinese, one sentence describing the design idea.
|
||||||
|
- negative_prompt: what to avoid (real people, likeness, characters, logos, text).
|
||||||
|
|
||||||
|
Return JSON with the field "designs" (array of objects with keys:
|
||||||
|
motif, art_style, color_palette, composition, concept, negative_prompt)."""
|
||||||
|
|
||||||
|
PINTEREST_ANALYZE_SCHEMA = {
|
||||||
|
"name": "pinterest_design_briefs",
|
||||||
|
"schema": {
|
||||||
|
"type": "object",
|
||||||
|
"properties": {
|
||||||
|
"designs": {
|
||||||
|
"type": "array",
|
||||||
|
"items": {
|
||||||
|
"type": "object",
|
||||||
|
"properties": {
|
||||||
|
"motif": {"type": "string"},
|
||||||
|
"art_style": {"type": "string"},
|
||||||
|
"color_palette": {"type": "string"},
|
||||||
|
"composition": {"type": "string"},
|
||||||
|
"concept": {"type": "string"},
|
||||||
|
"negative_prompt": {"type": "string"},
|
||||||
|
},
|
||||||
|
"required": ["motif", "art_style", "color_palette", "composition",
|
||||||
|
"concept", "negative_prompt"],
|
||||||
|
"additionalProperties": False,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"required": ["designs"],
|
||||||
|
"additionalProperties": False,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def build_pinterest_analyze_user_prompt(term: str, country: str, image_count: int) -> str:
|
||||||
|
return (
|
||||||
|
f"Country: {country}\n"
|
||||||
|
f"Pinterest search term: {term}\n"
|
||||||
|
f"Reference images attached: {image_count} images.\n\n"
|
||||||
|
f"Analyze the attached images and produce {image_count} ORIGINAL design briefs "
|
||||||
|
f"(one per image), each capturing the visual vibe as an original T-shirt print design. "
|
||||||
|
f"Do NOT copy the images."
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def call_openai_compatible(cfg, messages, timeout=90):
|
def call_openai_compatible(cfg, messages, timeout=90):
|
||||||
base_url = str(cfg.get("base_url", "https://api.openai.com/v1")).rstrip("/")
|
base_url = str(cfg.get("base_url", "https://api.openai.com/v1")).rstrip("/")
|
||||||
api_key = cfg.get("api_key", "")
|
api_key = cfg.get("api_key", "")
|
||||||
@@ -251,6 +363,41 @@ def call_openai_compatible(cfg, messages, timeout=90):
|
|||||||
return data["choices"][0]["message"]["content"]
|
return data["choices"][0]["message"]["content"]
|
||||||
|
|
||||||
|
|
||||||
|
def call_openai_compatible_structured(cfg, messages, json_schema, timeout=120):
|
||||||
|
"""调用 LLM 并返回结构化 JSON 文本。
|
||||||
|
|
||||||
|
优先 json_schema(strict 结构化输出);部分兼容厂商不支持 json_schema 时
|
||||||
|
自动回退 json_object(仍要求 JSON)。最终解析交给 _extract_json 兜底。
|
||||||
|
"""
|
||||||
|
base_url = str(cfg.get("base_url", "https://api.openai.com/v1")).rstrip("/")
|
||||||
|
api_key = cfg.get("api_key", "")
|
||||||
|
model = cfg.get("model", "gpt-4o-mini")
|
||||||
|
url = f"{base_url}/chat/completions"
|
||||||
|
headers = {"Authorization": f"Bearer {api_key}", "Content-Type": "application/json"}
|
||||||
|
payload = {
|
||||||
|
"model": model,
|
||||||
|
"messages": messages,
|
||||||
|
"temperature": float(cfg.get("temperature", 0.6)),
|
||||||
|
"response_format": {
|
||||||
|
"type": "json_schema",
|
||||||
|
"json_schema": {
|
||||||
|
"name": json_schema.get("name", "structured_output"),
|
||||||
|
"strict": True,
|
||||||
|
"schema": json_schema.get("schema", json_schema),
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
try:
|
||||||
|
resp = requests.post(url, json=payload, headers=headers, timeout=timeout, proxies=NO_PROXY)
|
||||||
|
resp.raise_for_status()
|
||||||
|
return resp.json()["choices"][0]["message"]["content"]
|
||||||
|
except Exception: # noqa: BLE001 兼容厂商不支持 json_schema → 回退 json_object
|
||||||
|
payload["response_format"] = {"type": "json_object"}
|
||||||
|
resp = requests.post(url, json=payload, headers=headers, timeout=timeout, proxies=NO_PROXY)
|
||||||
|
resp.raise_for_status()
|
||||||
|
return resp.json()["choices"][0]["message"]["content"]
|
||||||
|
|
||||||
|
|
||||||
def _retry(func, max_attempts=4, base_delay=4):
|
def _retry(func, max_attempts=4, base_delay=4):
|
||||||
last = None
|
last = None
|
||||||
for attempt in range(max_attempts):
|
for attempt in range(max_attempts):
|
||||||
@@ -345,6 +492,124 @@ class OpenAICompatBackend(LLMBackend):
|
|||||||
_cache_set(cache_key, out)
|
_cache_set(cache_key, out)
|
||||||
return out
|
return out
|
||||||
|
|
||||||
|
def generate_pinterest_terms(self, context: Dict[str, Any]) -> Dict[str, Any]:
|
||||||
|
"""生成 Pinterest 搜索词(json_schema 结构化 + 动态注入已用词防重复)。
|
||||||
|
|
||||||
|
context 字段:country, seeds, used_terms, count。
|
||||||
|
返回 {"search_terms": [str]};失败抛异常由节点兜底(回退种子词)。
|
||||||
|
"""
|
||||||
|
cfg = self._cfg
|
||||||
|
# 防御性上限:已用词最多注入 100 个,防 token 超限(节点层已截断,这里双保险)
|
||||||
|
ctx = dict(context or {})
|
||||||
|
used = [str(u) for u in (ctx.get("used_terms") or []) if str(u)]
|
||||||
|
max_used = int((cfg or {}).get("max_used_terms_in_prompt", 100) or 100)
|
||||||
|
if max_used > 0:
|
||||||
|
ctx["used_terms"] = used[-max_used:]
|
||||||
|
messages = [
|
||||||
|
{"role": "system", "content": PINTEREST_TERM_SYSTEM_PROMPT},
|
||||||
|
{"role": "user", "content": build_pinterest_term_user_prompt(ctx)},
|
||||||
|
]
|
||||||
|
raw = _retry(lambda: call_openai_compatible_structured(cfg, messages, PINTEREST_TERM_SCHEMA, timeout=120))
|
||||||
|
parsed = _extract_json(raw)
|
||||||
|
terms = [str(x).strip() for x in (parsed.get("search_terms", []) or []) if str(x).strip()]
|
||||||
|
return {"search_terms": terms}
|
||||||
|
|
||||||
|
def analyze_pinterest_images(self, image_paths: List[str], term: str, country: str = "") -> List[Dict[str, Any]]:
|
||||||
|
"""多模态分析 Pinterest 图片 → 原创设计简报列表。
|
||||||
|
|
||||||
|
图片输入不被模型支持(纯文本模型 400)时自动降级为纯文本分析(仅用搜索词)。
|
||||||
|
失败返回 [],由节点兜底(回退 mock 规则简报)。
|
||||||
|
"""
|
||||||
|
cfg = self._cfg
|
||||||
|
api_key = cfg.get("api_key", "")
|
||||||
|
if not api_key:
|
||||||
|
print("[pinterest_analyze] 未配置 LLM api_key,跳过图片分析")
|
||||||
|
return []
|
||||||
|
base_url = str(cfg.get("base_url") or "https://api.openai.com/v1").rstrip("/")
|
||||||
|
model = cfg.get("model", "gpt-4o-mini")
|
||||||
|
url = f"{base_url}/chat/completions"
|
||||||
|
headers = {"Authorization": f"Bearer {api_key}", "Content-Type": "application/json"}
|
||||||
|
|
||||||
|
# 图片 → base64 data URI(多模态输入)
|
||||||
|
data_uris: List[str] = []
|
||||||
|
for p in image_paths:
|
||||||
|
try:
|
||||||
|
import base64 as b64
|
||||||
|
mime = "image/png"
|
||||||
|
if Path(p).suffix.lower() in (".jpg", ".jpeg"):
|
||||||
|
mime = "image/jpeg"
|
||||||
|
data_uris.append(f"data:{mime};base64,{b64.b64encode(Path(p).read_bytes()).decode()}")
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
print(f"[pinterest_analyze] 图片读取失败 {p}: {e}")
|
||||||
|
|
||||||
|
def _call(use_images: bool) -> str:
|
||||||
|
user_content: List[Any] = [
|
||||||
|
{"type": "text", "text": build_pinterest_analyze_user_prompt(term, country, len(data_uris))},
|
||||||
|
]
|
||||||
|
if use_images:
|
||||||
|
user_content += [{"type": "image_url", "image_url": {"url": u}} for u in data_uris]
|
||||||
|
payload = {
|
||||||
|
"model": model,
|
||||||
|
"messages": [
|
||||||
|
{"role": "system", "content": PINTEREST_ANALYZE_SYSTEM_PROMPT},
|
||||||
|
{"role": "user", "content": user_content},
|
||||||
|
],
|
||||||
|
"temperature": 0.5,
|
||||||
|
"response_format": {
|
||||||
|
"type": "json_schema",
|
||||||
|
"json_schema": {
|
||||||
|
"name": PINTEREST_ANALYZE_SCHEMA["name"],
|
||||||
|
"strict": True,
|
||||||
|
"schema": PINTEREST_ANALYZE_SCHEMA["schema"],
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
try:
|
||||||
|
resp = requests.post(url, json=payload, headers=headers, timeout=180, proxies=NO_PROXY)
|
||||||
|
resp.raise_for_status()
|
||||||
|
return str(resp.json()["choices"][0]["message"].get("content") or "")
|
||||||
|
except Exception: # noqa: BLE001 兼容厂商不支持 json_schema
|
||||||
|
payload["response_format"] = {"type": "json_object"}
|
||||||
|
resp = requests.post(url, json=payload, headers=headers, timeout=180, proxies=NO_PROXY)
|
||||||
|
resp.raise_for_status()
|
||||||
|
return str(resp.json()["choices"][0]["message"].get("content") or "")
|
||||||
|
|
||||||
|
raw = ""
|
||||||
|
if data_uris:
|
||||||
|
try:
|
||||||
|
raw = _call(use_images=True)
|
||||||
|
except Exception as e: # noqa: BLE001 纯文本模型不支持图片 → 降级纯文本
|
||||||
|
print(f"[pinterest_analyze] 图片输入失败,降级纯文本分析: {e}")
|
||||||
|
raw = ""
|
||||||
|
if not raw:
|
||||||
|
try:
|
||||||
|
raw = _call(use_images=False)
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
print(f"[pinterest_analyze] 分析失败: {e}")
|
||||||
|
return []
|
||||||
|
try:
|
||||||
|
parsed = _extract_json(raw)
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
print(f"[pinterest_analyze] 解析失败: {e}")
|
||||||
|
return []
|
||||||
|
designs = []
|
||||||
|
for i, d in enumerate(parsed.get("designs") or []):
|
||||||
|
if not isinstance(d, dict):
|
||||||
|
continue
|
||||||
|
designs.append({
|
||||||
|
"topic": term,
|
||||||
|
"concept": str(d.get("concept", "")).strip(),
|
||||||
|
"motif": str(d.get("motif", "")).strip(),
|
||||||
|
"art_style": str(d.get("art_style", "")).strip(),
|
||||||
|
"color_palette": str(d.get("color_palette", "")).strip(),
|
||||||
|
"composition": str(d.get("composition", "")).strip(),
|
||||||
|
"negative_prompt": str(d.get("negative_prompt", "")).strip(),
|
||||||
|
# 生图参考:每条简报对应其来源爬取图(LLM 按图逐张产出简报,顺序一一对应)
|
||||||
|
"ref_images": [str(image_paths[i])] if i < len(image_paths) else [],
|
||||||
|
"source": "pinterest",
|
||||||
|
})
|
||||||
|
return designs
|
||||||
|
|
||||||
def generate_title(self, image_path: str, system_prompt: str = "", country: str = "",
|
def generate_title(self, image_path: str, system_prompt: str = "", country: str = "",
|
||||||
fallback_text: str = "") -> Dict[str, Any]:
|
fallback_text: str = "") -> Dict[str, Any]:
|
||||||
"""多模态标题生成;图片输入不被模型支持(如 qwen 纯文本模型 400)时,
|
"""多模态标题生成;图片输入不被模型支持(如 qwen 纯文本模型 400)时,
|
||||||
|
|||||||
@@ -3,6 +3,9 @@ from .compose_node import compose_node
|
|||||||
from .fetch_node import fetch_node
|
from .fetch_node import fetch_node
|
||||||
from .filter_node import filter_node
|
from .filter_node import filter_node
|
||||||
from .oss_upload_node import oss_upload_node
|
from .oss_upload_node import oss_upload_node
|
||||||
|
from .pinterest_analyze_node import pinterest_analyze_node
|
||||||
|
from .pinterest_scrape_node import pinterest_scrape_node
|
||||||
|
from .pinterest_search_node import pinterest_search_node
|
||||||
from .product_node import product_node
|
from .product_node import product_node
|
||||||
from .prompt_node import prompt_node
|
from .prompt_node import prompt_node
|
||||||
from .score_node import score_node
|
from .score_node import score_node
|
||||||
@@ -23,4 +26,7 @@ __all__ = [
|
|||||||
"oss_upload_node",
|
"oss_upload_node",
|
||||||
"seed_shot_node",
|
"seed_shot_node",
|
||||||
"template_export_node",
|
"template_export_node",
|
||||||
|
"pinterest_search_node",
|
||||||
|
"pinterest_scrape_node",
|
||||||
|
"pinterest_analyze_node",
|
||||||
]
|
]
|
||||||
|
|||||||
@@ -157,15 +157,38 @@ def compose_node(state: Dict[str, Any]) -> Dict[str, Any]:
|
|||||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||||
|
|
||||||
def _gen_one(i: int, b: Dict[str, Any]):
|
def _gen_one(i: int, b: Dict[str, Any]):
|
||||||
"""单张设计稿生成(并发线程内调用,每设计一线程)。"""
|
"""单张设计稿生成(并发线程内调用,每设计一线程)。
|
||||||
|
|
||||||
|
Pinterest 参考模式:简报带 ref_images(爬取图)→ 用 ib.print() 图生图,
|
||||||
|
把爬取图 + 多模态分析简报(已封装进 image_prompt)一起发给生图模型;
|
||||||
|
无参考图或图生图失败 → 回退 ib.generate() 纯文生图。
|
||||||
|
"""
|
||||||
try:
|
try:
|
||||||
img_prompt = sanitize_image_prompt(b.get("image_prompt", ""))
|
img_prompt = sanitize_image_prompt(b.get("image_prompt", ""))
|
||||||
img_prompt = ensure_rebrand_hint(b, img_prompt) # review → 原创化魔改引导
|
img_prompt = ensure_rebrand_hint(b, img_prompt) # review → 原创化魔改引导
|
||||||
out_path = ib.generate(
|
out_path = str(design_dir / f"{country}_{i:02d}_design.png")
|
||||||
img_prompt,
|
ref_images = [str(p) for p in (b.get("ref_images") or []) if str(p)]
|
||||||
str(design_dir / f"{country}_{i:02d}_design.png"),
|
if ref_images and hasattr(ib, "print"):
|
||||||
|
try:
|
||||||
|
# 图生图:以爬取图为参考,按分析简报生成原创设计(不复制原图)
|
||||||
|
ref_prompt = img_prompt + (
|
||||||
|
" Create an ORIGINAL, non-copying flat print design inspired ONLY by "
|
||||||
|
"the reference image's style and mood. Do NOT reproduce the reference "
|
||||||
|
"image, its characters, logos, or any text.")
|
||||||
|
out_path = ib.print(
|
||||||
|
ref_prompt, ref_images[0], out_path,
|
||||||
b.get("composite_negative", ""),
|
b.get("composite_negative", ""),
|
||||||
|
extra_images=ref_images[1:] or None,
|
||||||
size="1024x1024") # 印花设计统一 1024x1024
|
size="1024x1024") # 印花设计统一 1024x1024
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
print(f"[compose] 图生图(参考图)失败,回退文生图 {b.get('topic','')}: {e}")
|
||||||
|
out_path = ib.generate(
|
||||||
|
img_prompt, str(design_dir / f"{country}_{i:02d}_design.png"),
|
||||||
|
b.get("composite_negative", ""), size="1024x1024")
|
||||||
|
else:
|
||||||
|
out_path = ib.generate(
|
||||||
|
img_prompt, str(design_dir / f"{country}_{i:02d}_design.png"),
|
||||||
|
b.get("composite_negative", ""), size="1024x1024")
|
||||||
return i, b, out_path, None
|
return i, b, out_path, None
|
||||||
except Exception as e: # noqa: BLE001
|
except Exception as e: # noqa: BLE001
|
||||||
return i, b, None, e
|
return i, b, None, e
|
||||||
|
|||||||
@@ -0,0 +1,148 @@
|
|||||||
|
"""Pinterest 参考模式节点 3/3:LLM 多模态分析图片 → 原创设计简报(pinterest_analyze)。
|
||||||
|
|
||||||
|
对 pinterest_scrape 爬到的每个搜索词图片,调 LLM 多模态分析(analyze_pinterest_images)
|
||||||
|
提取视觉概念(风格/情绪/主体/配色/构图)→ 生成原创设计简报
|
||||||
|
(motif/art_style/color_palette/composition/concept/negative_prompt),
|
||||||
|
再经 prompt_node 装配最终 image/wearable/composite 提示词,产出标准 briefs 供 compose 用。
|
||||||
|
|
||||||
|
兜底链:LLM 多模态 → 纯文本降级(后端内部)→ mock 规则简报 → 空列表(下游跳过)。
|
||||||
|
带 with_fallback:任何异常都不中断。
|
||||||
|
"""
|
||||||
|
from typing import Any, Dict, List
|
||||||
|
|
||||||
|
from graph.llms import get_backend
|
||||||
|
from graph.nodes.prompt_node import prompt_node
|
||||||
|
from graph.validate import with_fallback
|
||||||
|
|
||||||
|
|
||||||
|
def _enrich_briefs(raw_briefs: List[Dict[str, Any]], country: str) -> List[Dict[str, Any]]:
|
||||||
|
"""富化原始简报 → screened 格式(唯一 topic / safe / 分类 / 分数),供 prompt_node 装配。
|
||||||
|
|
||||||
|
同一搜索词的多张图会产出多条简报,topic 相同 → 追加序号保证唯一
|
||||||
|
(product_node 按 topic 绑定简报,重复 topic 会互相覆盖)。
|
||||||
|
"""
|
||||||
|
from graph.classify import classify
|
||||||
|
seen_topics: Dict[str, int] = {}
|
||||||
|
out: List[Dict[str, Any]] = []
|
||||||
|
for i, b in enumerate(raw_briefs):
|
||||||
|
if not isinstance(b, dict):
|
||||||
|
continue
|
||||||
|
term = str(b.get("topic") or "").strip() or f"pinterest {i + 1}"
|
||||||
|
base = term
|
||||||
|
n = seen_topics.get(base.lower(), 0)
|
||||||
|
seen_topics[base.lower()] = n + 1
|
||||||
|
topic = base if n == 0 else f"{base} #{n + 1}"
|
||||||
|
motif = str(b.get("motif") or "").strip() or term
|
||||||
|
if not motif:
|
||||||
|
continue
|
||||||
|
out.append({
|
||||||
|
"country": country,
|
||||||
|
"topic": topic,
|
||||||
|
"risk_level": "safe",
|
||||||
|
"safe_for_print": True,
|
||||||
|
"suitable_for_print": True,
|
||||||
|
"design_category": classify(term),
|
||||||
|
"concept": str(b.get("concept") or "").strip() or f"围绕「{term}」的原创印花设计",
|
||||||
|
"motif": motif,
|
||||||
|
"art_style": str(b.get("art_style") or "").strip(),
|
||||||
|
"color_palette": str(b.get("color_palette") or "").strip(),
|
||||||
|
"composition": str(b.get("composition") or "").strip(),
|
||||||
|
"negative_prompt": str(b.get("negative_prompt") or "").strip(),
|
||||||
|
"ref_images": [str(p) for p in (b.get("ref_images") or []) if str(p)],
|
||||||
|
"slogan": "",
|
||||||
|
"score": 1.0,
|
||||||
|
"confidence": 1.0,
|
||||||
|
"source": "pinterest",
|
||||||
|
})
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
@with_fallback("pinterest_analyze")
|
||||||
|
def pinterest_analyze_node(state: Dict[str, Any]) -> Dict[str, Any]:
|
||||||
|
images: Dict[str, List[str]] = state.get("pinterest_images") or {}
|
||||||
|
if not images:
|
||||||
|
print("[pinterest_analyze] 无爬取图片,跳过分析")
|
||||||
|
return {"pinterest_briefs": [], "briefs": [], "errors": state.get("errors") or []}
|
||||||
|
|
||||||
|
country = state["country"]
|
||||||
|
config = state["config"]
|
||||||
|
errors = list(state.get("errors") or [])
|
||||||
|
|
||||||
|
pcfg = config.get("pinterest") or {}
|
||||||
|
analyze_per_term = int(pcfg.get("analyze_per_term", 6))
|
||||||
|
max_designs = int(pcfg.get("max_designs", 10))
|
||||||
|
provider = str(pcfg.get("provider") or "openai").strip().lower()
|
||||||
|
|
||||||
|
# 1) LLM 后端(openai → 真多模态;mock → 规则兜底)
|
||||||
|
llm = None
|
||||||
|
if provider != "static":
|
||||||
|
try:
|
||||||
|
llm = get_backend(provider)
|
||||||
|
if hasattr(llm, "bind_config"):
|
||||||
|
llm.bind_config(config.get("llm_screen") or {})
|
||||||
|
if provider not in ("mock",) and not getattr(llm, "has_key", False):
|
||||||
|
print(f"[pinterest_analyze] {provider} 未配置 API key,降级 mock")
|
||||||
|
llm = get_backend("mock")
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
print(f"[pinterest_analyze] LLM 初始化失败: {e}")
|
||||||
|
llm = None
|
||||||
|
|
||||||
|
# 2) 逐搜索词分析图片 → 原始设计简报
|
||||||
|
raw_briefs: List[Dict[str, Any]] = []
|
||||||
|
if llm is not None and hasattr(llm, "analyze_pinterest_images"):
|
||||||
|
for term, paths in images.items():
|
||||||
|
sample = list(paths)[:analyze_per_term]
|
||||||
|
if not sample:
|
||||||
|
continue
|
||||||
|
try:
|
||||||
|
res = llm.analyze_pinterest_images(sample, term, country)
|
||||||
|
res = res or []
|
||||||
|
raw_briefs.extend(res)
|
||||||
|
print(f"[pinterest_analyze] 「{term}」分析 {len(sample)} 张图 → {len(res)} 条简报")
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
errors.append({"node": "pinterest_analyze", "type": type(e).__name__,
|
||||||
|
"message": f"term[{term}]: {e}", "trace": ""})
|
||||||
|
print(f"[pinterest_analyze] 「{term}」分析失败: {e}")
|
||||||
|
|
||||||
|
# 3) 兜底:LLM 无结果 → mock 规则简报(零 API 成本,保证有设计可生成)
|
||||||
|
if not raw_briefs and llm is not None:
|
||||||
|
try:
|
||||||
|
for term, paths in images.items():
|
||||||
|
sample = list(paths)[:analyze_per_term]
|
||||||
|
if sample:
|
||||||
|
raw_briefs.extend(llm.analyze_pinterest_images(sample, term, country) or [])
|
||||||
|
print(f"[pinterest_analyze] 兜底:mock 规则简报 {len(raw_briefs)} 条")
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
print(f"[pinterest_analyze] mock 兜底失败: {e}")
|
||||||
|
|
||||||
|
# 4) 上限 + 去重(同 motif+style 指纹只留一条)
|
||||||
|
raw_briefs = raw_briefs[:max_designs]
|
||||||
|
seen: set = set()
|
||||||
|
uniq: List[Dict[str, Any]] = []
|
||||||
|
for b in raw_briefs:
|
||||||
|
if not isinstance(b, dict):
|
||||||
|
continue
|
||||||
|
fp = f"{str(b.get('motif', '')).strip().lower()}|{str(b.get('art_style', '')).strip().lower()}"
|
||||||
|
if fp in seen:
|
||||||
|
continue
|
||||||
|
seen.add(fp)
|
||||||
|
uniq.append(b)
|
||||||
|
raw_briefs = uniq
|
||||||
|
|
||||||
|
# 5) 富化 → screened → prompt_node 装配提示词 → 标准 briefs
|
||||||
|
screened = _enrich_briefs(raw_briefs, country)
|
||||||
|
if not screened:
|
||||||
|
print("[pinterest_analyze] 无有效设计简报,跳过")
|
||||||
|
return {"pinterest_briefs": [], "briefs": [], "errors": errors}
|
||||||
|
|
||||||
|
r = prompt_node({**state, "screened": screened})
|
||||||
|
briefs = r.get("briefs") or []
|
||||||
|
|
||||||
|
stats = dict(state.get("stats") or {})
|
||||||
|
stats["pinterest_analyze"] = {
|
||||||
|
"provider": provider,
|
||||||
|
"images_analyzed": sum(len(v) for v in images.values()),
|
||||||
|
"briefs": len(briefs),
|
||||||
|
}
|
||||||
|
print(f"[pinterest_analyze] 设计简报 {len(briefs)} 条({country})")
|
||||||
|
return {"pinterest_briefs": raw_briefs, "briefs": briefs, "stats": stats, "errors": errors}
|
||||||
@@ -0,0 +1,80 @@
|
|||||||
|
"""Pinterest 参考模式节点 2/3:爬取图片(pinterest_scrape)。
|
||||||
|
|
||||||
|
对 pinterest_search 生成的每个搜索词,调 pinterest_scraper.scraper.scrape_pinterest
|
||||||
|
(Playwright 启动本地 Chrome)搜索 Pinterest 并下载图片到
|
||||||
|
output/pinterest_ref/<国家>/<搜索词>/。
|
||||||
|
|
||||||
|
- 单个搜索词失败(未登录/网络/无结果)跳过,不中断整批。
|
||||||
|
- 并发数由 config.pinterest.scrape_concurrency 控制(每个并发开一个 Chrome 窗口)。
|
||||||
|
- 已爬取过且图片数达标的搜索词跳过(断点续爬,避免重复开 Chrome)。
|
||||||
|
"""
|
||||||
|
import concurrent.futures
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any, Dict, List
|
||||||
|
|
||||||
|
from graph.validate import with_fallback
|
||||||
|
|
||||||
|
|
||||||
|
def _term_dir(output_dir: str, country: str, term: str) -> Path:
|
||||||
|
safe = "".join(ch for ch in term if ch.isalnum() or ch in "-_ ").strip() or "term"
|
||||||
|
return Path(output_dir) / "pinterest_ref" / country / safe
|
||||||
|
|
||||||
|
|
||||||
|
def _already_scraped(term_dir: Path) -> bool:
|
||||||
|
"""该搜索词已爬取过(目录里已有 ≥1 张图)→ 跳过,避免重复开 Chrome。"""
|
||||||
|
if not term_dir.exists():
|
||||||
|
return False
|
||||||
|
return any(p.is_file() and p.suffix.lower() in (".jpg", ".jpeg", ".png", ".webp")
|
||||||
|
for p in term_dir.iterdir())
|
||||||
|
|
||||||
|
|
||||||
|
@with_fallback("pinterest_scrape")
|
||||||
|
def pinterest_scrape_node(state: Dict[str, Any]) -> Dict[str, Any]:
|
||||||
|
terms: List[str] = state.get("pinterest_search_terms") or []
|
||||||
|
if not terms:
|
||||||
|
print("[pinterest_scrape] 无搜索词,跳过爬取")
|
||||||
|
return {"pinterest_images": {}, "errors": state.get("errors") or []}
|
||||||
|
|
||||||
|
country = state["country"]
|
||||||
|
config = state["config"]
|
||||||
|
output_dir = state["output_dir"]
|
||||||
|
errors = list(state.get("errors") or [])
|
||||||
|
|
||||||
|
pcfg = config.get("pinterest") or {}
|
||||||
|
images_per_term = int(pcfg.get("images_per_term", 40))
|
||||||
|
concurrency = int(pcfg.get("scrape_concurrency", 2))
|
||||||
|
headless = bool(pcfg.get("headless", False))
|
||||||
|
proxy = pcfg.get("proxy") or None
|
||||||
|
|
||||||
|
results: Dict[str, List[str]] = {}
|
||||||
|
skipped: List[str] = []
|
||||||
|
|
||||||
|
def _one(term: str) -> None:
|
||||||
|
term_dir = _term_dir(output_dir, country, term)
|
||||||
|
if _already_scraped(term_dir):
|
||||||
|
skipped.append(term)
|
||||||
|
print(f"[pinterest_scrape] 已爬取过(跳过): {term}")
|
||||||
|
return
|
||||||
|
try:
|
||||||
|
from pinterest_scraper.scraper import scrape_pinterest
|
||||||
|
files = scrape_pinterest(term, count=images_per_term,
|
||||||
|
save_dir=str(term_dir), proxy=proxy, headless=headless)
|
||||||
|
results[term] = files
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
errors.append({"node": "pinterest_scrape", "type": type(e).__name__,
|
||||||
|
"message": f"term[{term}]: {e}", "trace": ""})
|
||||||
|
print(f"[pinterest_scrape] 爬取失败(跳过): {term}: {e}")
|
||||||
|
|
||||||
|
print(f"[pinterest_scrape] 开始爬取 {len(terms)} 个搜索词(并发 {concurrency})…")
|
||||||
|
with concurrent.futures.ThreadPoolExecutor(max_workers=max(1, concurrency)) as ex:
|
||||||
|
list(ex.map(_one, terms))
|
||||||
|
|
||||||
|
total = sum(len(v) for v in results.values())
|
||||||
|
stats = dict(state.get("stats") or {})
|
||||||
|
stats["pinterest_scrape"] = {
|
||||||
|
"terms": len(terms), "scraped": len(results), "skipped": len(skipped),
|
||||||
|
"images": total,
|
||||||
|
}
|
||||||
|
print(f"[pinterest_scrape] 完成:{len(results)} 个搜索词,共 {total} 张图(跳过 {len(skipped)})")
|
||||||
|
|
||||||
|
return {"pinterest_images": results, "stats": stats, "errors": errors}
|
||||||
@@ -0,0 +1,103 @@
|
|||||||
|
"""Pinterest 参考模式节点 1/3:LLM 生成搜索词(pinterest_search)。
|
||||||
|
|
||||||
|
流程:国家 Pinterest 种子词池 → LLM 生成搜索词(json_schema 结构化 + 动态注入已用词防重复)
|
||||||
|
→ 全局过滤(已用/黑名单/不适合T恤/去重)→ 持久化已用词。
|
||||||
|
|
||||||
|
兜底链:LLM json_schema → json_object → 解析失败/调用失败 → 回退种子词池随机抽样。
|
||||||
|
带 with_fallback:任何异常都不中断,返回空列表由下游跳过。
|
||||||
|
"""
|
||||||
|
import random
|
||||||
|
from typing import Any, Dict, List
|
||||||
|
|
||||||
|
from graph.llms import get_backend
|
||||||
|
from graph.pinterest import (
|
||||||
|
filter_search_terms,
|
||||||
|
load_used_terms,
|
||||||
|
merge_used,
|
||||||
|
sample_seeds,
|
||||||
|
save_used_terms,
|
||||||
|
)
|
||||||
|
from graph.validate import with_fallback
|
||||||
|
|
||||||
|
|
||||||
|
@with_fallback("pinterest_search")
|
||||||
|
def pinterest_search_node(state: Dict[str, Any]) -> Dict[str, Any]:
|
||||||
|
country = state["country"]
|
||||||
|
config = state["config"]
|
||||||
|
output_dir = state["output_dir"]
|
||||||
|
errors = list(state.get("errors") or [])
|
||||||
|
|
||||||
|
pcfg = config.get("pinterest") or {}
|
||||||
|
if not pcfg.get("enabled", True):
|
||||||
|
return {"pinterest_search_terms": [], "errors": errors}
|
||||||
|
|
||||||
|
provider = str(pcfg.get("provider") or "openai").strip().lower()
|
||||||
|
want = int(pcfg.get("search_terms_per_run", 10))
|
||||||
|
seed_sample = int(pcfg.get("seed_sample", 40))
|
||||||
|
max_used_in_prompt = int(pcfg.get("max_used_terms_in_prompt", 100))
|
||||||
|
blacklist = config.get("blacklist") or []
|
||||||
|
|
||||||
|
# 1) 种子词池(随机抽样)+ 已用搜索词
|
||||||
|
seeds = sample_seeds(country, seed_sample)
|
||||||
|
used = load_used_terms(output_dir, country)
|
||||||
|
if not seeds:
|
||||||
|
print(f"[pinterest_search] {country} 无种子词,跳过搜索词生成")
|
||||||
|
return {"pinterest_search_terms": [], "errors": errors}
|
||||||
|
|
||||||
|
# 2) LLM 生成(json_schema + 动态注入已用词)
|
||||||
|
# 已用词只取最近 N 个(默认 100)注入提示词,防 token 超限;过滤仍用全量。
|
||||||
|
used_llm = used[-max_used_in_prompt:] if max_used_in_prompt > 0 else []
|
||||||
|
terms: List[str] = []
|
||||||
|
llm = None
|
||||||
|
if provider != "static":
|
||||||
|
try:
|
||||||
|
llm = get_backend(provider)
|
||||||
|
if hasattr(llm, "bind_config"):
|
||||||
|
llm.bind_config(config.get("llm_screen") or {})
|
||||||
|
if provider not in ("mock",) and not getattr(llm, "has_key", False):
|
||||||
|
print(f"[pinterest_search] {provider} 未配置 API key,降级 mock")
|
||||||
|
llm = get_backend("mock")
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
print(f"[pinterest_search] LLM 初始化失败: {e}")
|
||||||
|
llm = None
|
||||||
|
|
||||||
|
if llm is not None and hasattr(llm, "generate_pinterest_terms"):
|
||||||
|
try:
|
||||||
|
ctx = {"country": country, "seeds": seeds, "used_terms": used_llm, "count": want}
|
||||||
|
res = llm.generate_pinterest_terms(ctx)
|
||||||
|
terms = [str(t).strip() for t in (res.get("search_terms") or []) if str(t).strip()]
|
||||||
|
print(f"[pinterest_search] LLM 生成搜索词 {len(terms)} 个({country},已用词注入 {len(used_llm)}/{len(used)})")
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
print(f"[pinterest_search] LLM 生成失败,回退种子词池: {e}")
|
||||||
|
terms = []
|
||||||
|
|
||||||
|
# 3) 兜底:LLM 无结果 → 种子词池随机抽样
|
||||||
|
if not terms:
|
||||||
|
terms = random.sample(seeds, min(want, len(seeds))) if seeds else []
|
||||||
|
print(f"[pinterest_search] 兜底:从种子词池取 {len(terms)} 个")
|
||||||
|
|
||||||
|
# 4) 全局过滤(已用/黑名单/不适合T恤/去重)
|
||||||
|
filtered = filter_search_terms(terms, used, blacklist)
|
||||||
|
if len(filtered) < want and seeds:
|
||||||
|
# 不足时用种子词池补充(同样过滤),保证数量
|
||||||
|
extra = filter_search_terms(seeds, merge_used(used, filtered), blacklist)
|
||||||
|
for t in extra:
|
||||||
|
if len(filtered) >= want:
|
||||||
|
break
|
||||||
|
filtered.append(t)
|
||||||
|
|
||||||
|
# 5) 持久化已用词
|
||||||
|
new_used = merge_used(used, filtered)
|
||||||
|
save_used_terms(output_dir, country, new_used)
|
||||||
|
|
||||||
|
stats = dict(state.get("stats") or {})
|
||||||
|
stats["pinterest_search"] = {
|
||||||
|
"provider": provider,
|
||||||
|
"generated": len(terms),
|
||||||
|
"filtered": len(filtered),
|
||||||
|
"used_total": len(new_used),
|
||||||
|
}
|
||||||
|
print(f"[pinterest_search] 搜索词 {len(filtered)} 个(已用累计 {len(new_used)}): "
|
||||||
|
f"{', '.join(filtered[:6])}{'...' if len(filtered) > 6 else ''}")
|
||||||
|
|
||||||
|
return {"pinterest_search_terms": filtered, "stats": stats, "errors": errors}
|
||||||
@@ -0,0 +1,116 @@
|
|||||||
|
"""Pinterest 参考模式共享辅助:种子词加载、已用搜索词持久化、搜索词全局过滤。
|
||||||
|
|
||||||
|
独立于 Google Trends 采集链路,供 pinterest_search / scrape / analyze 节点复用。
|
||||||
|
"""
|
||||||
|
import json
|
||||||
|
import random
|
||||||
|
import re
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any, Dict, List
|
||||||
|
|
||||||
|
from graph.paths import project_root, runtime_root
|
||||||
|
|
||||||
|
# 不适合 T 恤印花的类目关键词(复用 product_batch 的兜底清单)
|
||||||
|
_UNSUITABLE = re.compile(
|
||||||
|
r"\b(nails?|manicure|pedicure|recipes?|cooking|lottery|jackpot|results?|score|scores?|"
|
||||||
|
r"fixtures?|forecast|weather|temperature|map|directions?|parking|opening hours?|"
|
||||||
|
r"prices?|price|reviews?|jobs?|salary|mortgage|council tax|election|referendum|"
|
||||||
|
r"stock market|exchange rate|gas prices?)\b",
|
||||||
|
re.IGNORECASE,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def pinterest_seed_path(country: str) -> Path:
|
||||||
|
for root in (runtime_root(), project_root()):
|
||||||
|
p = root / "configs" / "pinterest" / f"{country}.yaml"
|
||||||
|
if p.exists():
|
||||||
|
return p
|
||||||
|
return Path("configs") / "pinterest" / f"{country}.yaml"
|
||||||
|
|
||||||
|
|
||||||
|
def load_pinterest_seeds(country: str) -> List[str]:
|
||||||
|
"""读国家 Pinterest 种子词池(configs/pinterest/<CC>.yaml 的 seeds)。"""
|
||||||
|
try:
|
||||||
|
import yaml
|
||||||
|
p = pinterest_seed_path(country)
|
||||||
|
if not p.exists():
|
||||||
|
print(f"[pinterest] 未找到种子词配置: {p}")
|
||||||
|
return []
|
||||||
|
data = yaml.safe_load(p.read_text(encoding="utf-8")) or {}
|
||||||
|
seeds = [str(s).strip() for s in (data.get("seeds") or []) if str(s).strip()]
|
||||||
|
return seeds
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
print(f"[pinterest] 种子词加载失败: {e}")
|
||||||
|
return []
|
||||||
|
|
||||||
|
|
||||||
|
def sample_seeds(country: str, n: int) -> List[str]:
|
||||||
|
"""从国家种子池随机抽取 n 个种子词(不足则全取)。"""
|
||||||
|
seeds = load_pinterest_seeds(country)
|
||||||
|
if not seeds:
|
||||||
|
return []
|
||||||
|
if len(seeds) <= n:
|
||||||
|
return list(seeds)
|
||||||
|
return random.sample(seeds, n)
|
||||||
|
|
||||||
|
|
||||||
|
def used_terms_path(output_dir: str, country: str) -> Path:
|
||||||
|
return Path(output_dir) / "pinterest_ref" / country / "used_search_terms.json"
|
||||||
|
|
||||||
|
|
||||||
|
def load_used_terms(output_dir: str, country: str) -> List[str]:
|
||||||
|
"""读已用搜索词(跨多次运行持久化,供动态注入防重复)。"""
|
||||||
|
try:
|
||||||
|
p = used_terms_path(output_dir, country)
|
||||||
|
if p.exists():
|
||||||
|
data = json.loads(p.read_text(encoding="utf-8"))
|
||||||
|
return [str(t).strip() for t in (data.get("terms") or []) if str(t).strip()]
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
print(f"[pinterest] 已用搜索词读取失败: {e}")
|
||||||
|
return []
|
||||||
|
|
||||||
|
|
||||||
|
def save_used_terms(output_dir: str, country: str, terms: List[str]) -> None:
|
||||||
|
"""持久化已用搜索词(去重保序)。"""
|
||||||
|
try:
|
||||||
|
p = used_terms_path(output_dir, country)
|
||||||
|
p.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
seen, out = set(), []
|
||||||
|
for t in terms:
|
||||||
|
k = t.strip().lower()
|
||||||
|
if k and k not in seen:
|
||||||
|
seen.add(k)
|
||||||
|
out.append(t.strip())
|
||||||
|
p.write_text(json.dumps({"terms": out}, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
print(f"[pinterest] 已用搜索词保存失败: {e}")
|
||||||
|
|
||||||
|
|
||||||
|
def filter_search_terms(terms: List[str], used: List[str], blacklist: List[str]) -> List[str]:
|
||||||
|
"""全局搜索词过滤:剔除已用、黑名单、不适合 T 恤类目、去重(大小写不敏感)。"""
|
||||||
|
used_set = {str(u).strip().lower() for u in used if str(u).strip()}
|
||||||
|
black = [str(b).strip().lower() for b in (blacklist or []) if str(b).strip()]
|
||||||
|
seen, out = set(), []
|
||||||
|
for t in terms:
|
||||||
|
s = str(t).strip()
|
||||||
|
low = s.lower()
|
||||||
|
if not s or low in seen or low in used_set:
|
||||||
|
continue
|
||||||
|
if any(b and b in low for b in black):
|
||||||
|
continue
|
||||||
|
if _UNSUITABLE.search(low):
|
||||||
|
continue
|
||||||
|
seen.add(low)
|
||||||
|
out.append(s)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def merge_used(existing: List[str], new_terms: List[str]) -> List[str]:
|
||||||
|
"""合并已用搜索词(新词追加到末尾,去重保序)。"""
|
||||||
|
seen, out = set(), []
|
||||||
|
for t in list(existing) + list(new_terms):
|
||||||
|
k = str(t).strip().lower()
|
||||||
|
if k and k not in seen:
|
||||||
|
seen.add(k)
|
||||||
|
out.append(str(t).strip())
|
||||||
|
return out
|
||||||
@@ -29,6 +29,11 @@ class AgentState(TypedDict, total=False):
|
|||||||
product: List[Dict[str, Any]] # product 产出:产品图生成(SPU/SKU/底图/印花/模特合成)
|
product: List[Dict[str, Any]] # product 产出:产品图生成(SPU/SKU/底图/印花/模特合成)
|
||||||
seed_words: Dict[str, Any] # seed 产出:动态种子词(含 llm_style_seeds / llm_related_seeds)
|
seed_words: Dict[str, Any] # seed 产出:动态种子词(含 llm_style_seeds / llm_related_seeds)
|
||||||
|
|
||||||
|
# —— Pinterest 参考模式(独立于 Google Trends 采集链路)——
|
||||||
|
pinterest_search_terms: List[str] # pinterest_search 产出:LLM 生成的搜索词
|
||||||
|
pinterest_images: Dict[str, List[str]] # pinterest_scrape 产出:搜索词 → 爬取图片路径列表
|
||||||
|
pinterest_briefs: List[Dict[str, Any]] # pinterest_analyze 产出:LLM 分析图片的原始设计简报
|
||||||
|
|
||||||
# —— 可观测性 ——
|
# —— 可观测性 ——
|
||||||
errors: List[Dict[str, Any]] # 各节点兜底捕获的错误:{node, type, message, trace}
|
errors: List[Dict[str, Any]] # 各节点兜底捕获的错误:{node, type, message, trace}
|
||||||
stats: Dict[str, Any] # 各阶段统计:{fetch, filter, score, screen, prompt, compose}
|
stats: Dict[str, Any] # 各阶段统计:{fetch, filter, score, screen, prompt, compose}
|
||||||
|
|||||||
@@ -0,0 +1,76 @@
|
|||||||
|
# Pinterest 关键词图片爬取工具
|
||||||
|
|
||||||
|
用 Playwright 启动本机 **Google Chrome** 搜索 Pinterest,并下载最高分辨率的图片。登录态通过**项目内持久化缓存目录**自动保存,首次登录后无需重复操作。
|
||||||
|
|
||||||
|
## 前置条件
|
||||||
|
|
||||||
|
1. 安装 Python 依赖:`uv sync`
|
||||||
|
2. 本机已安装 **Google Chrome**(脚本直接复用它,不下载任何浏览器内核)
|
||||||
|
|
||||||
|
## 运行方式
|
||||||
|
|
||||||
|
### 最简用法(默认,会自动缓存登录态)
|
||||||
|
|
||||||
|
```bash
|
||||||
|
uv run pinterest_image_capture.py "风景壁纸" -n 50
|
||||||
|
```
|
||||||
|
|
||||||
|
脚本会用项目内的 `.chrome_session/User Data` 作为 Chrome 用户数据目录启动一个**可见的本地 Chrome 窗口**:
|
||||||
|
|
||||||
|
- **首次运行**:请在弹出的窗口里登录 Pinterest(只需这一次)。登录态会缓存在 `.chrome_session` 目录中。
|
||||||
|
- **后续运行**:直接复用缓存的登录态,自动登录,不再需要手动登录。
|
||||||
|
- 你能实时看到滚动与采集过程,脚本结束自动关窗。
|
||||||
|
- `.chrome_session` 已加入 `.gitignore`,不会误提交。
|
||||||
|
|
||||||
|
### 模式 B:接管你已开好的调试版主浏览器(可选)
|
||||||
|
|
||||||
|
如果你平时就用 `open_chrome_debug.bat`(或桌面快捷方式)开着调试版主浏览器,可加 `--cdp` 直接接管,跳过自动启动:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
uv run pinterest_image_capture.py "风景壁纸" -n 50 --cdp http://127.0.0.1:9222
|
||||||
|
```
|
||||||
|
|
||||||
|
## 代理
|
||||||
|
|
||||||
|
代理按以下顺序解析,无需手动填:命令行 `--proxy` > 环境变量 > 本地常见端口探测 > **系统代理**(兜底)。你机器的系统代理会被自动读取并同时用于浏览器与图片下载。传 `--proxy ""` 可禁用。
|
||||||
|
|
||||||
|
### 参数
|
||||||
|
|
||||||
|
| 参数 | 说明 | 默认值 |
|
||||||
|
|------|------|--------|
|
||||||
|
| `keyword` | 搜索关键词(必填,位置参数) | - |
|
||||||
|
| `-n, --count` | 爬取图片数量 | 40 |
|
||||||
|
| `--proxy` | 代理地址(默认自动解析:本地探测/系统代理;传空字符串禁用) | 自动 |
|
||||||
|
| `--cdp` | 可选:指定 CDP 地址以接管已开的调试版 Chrome(见模式 B) | 不填(默认用持久化缓存目录启动本地 Chrome) |
|
||||||
|
| `--headless` | 自动启动 Chrome 时采用无头模式 | 关闭 |
|
||||||
|
| `-o, --output` | 保存目录 | `output/img/<关键词>` |
|
||||||
|
|
||||||
|
### 示例
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# 最简用法:只填关键词和数量,代理自动探测
|
||||||
|
uv run pinterest_image_capture.py "风景壁纸" -n 50
|
||||||
|
|
||||||
|
# 无头模式运行(不弹窗)
|
||||||
|
uv run pinterest_image_capture.py 猫 -n 10 --headless
|
||||||
|
|
||||||
|
# 手动指定代理 / 禁用代理
|
||||||
|
uv run pinterest_image_capture.py 猫 -n 10 --proxy "http://127.0.0.1:7890"
|
||||||
|
uv run pinterest_image_capture.py 猫 -n 10 --proxy ""
|
||||||
|
```
|
||||||
|
|
||||||
|
## 代理自动探测
|
||||||
|
|
||||||
|
脚本会按以下顺序自动确定代理,无需手动填写:
|
||||||
|
|
||||||
|
1. 读取系统/用户环境变量 `HTTPS_PROXY` / `HTTP_PROXY`(含小写)。
|
||||||
|
2. 实测本地常见代理端口:Clash(`7890~7893`)、v2rayN(`10808`/`10809`)、Shadowsocks(`1080`/`1087`)、Fiddler(`8888`)等。
|
||||||
|
|
||||||
|
浏览器和下载阶段都会走同一个代理。若都没探测到,脚本会直连运行并在控制台提示(此时若无法访问 Pinterest 需手动用 `--proxy` 指定)。
|
||||||
|
|
||||||
|
## 说明
|
||||||
|
|
||||||
|
- 连续 10 轮滚动无新图片时自动停止(搜索结果不足目标数量时不会死循环)。
|
||||||
|
- 下载阶段并发数为 10,浏览器与下载都走自动探测到的代理。
|
||||||
|
- 默认模式下脚本启动的 Chrome 用 `.chrome_session` 用户目录,登录态持久化保存;脚本结束自动关窗。如需清空登录态,删除 `.chrome_session` 目录即可重新登录。
|
||||||
|
- `--cdp` 接管模式下脚本只断开连接,不会关闭你的主浏览器。
|
||||||
@@ -0,0 +1,25 @@
|
|||||||
|
@echo off
|
||||||
|
REM 以调试模式打开一个常驻的 Chrome 窗口,供脚本通过 --cdp 接管。
|
||||||
|
REM 使用项目内 .chrome_session 目录(非默认位置,Chrome 才允许开启远程调试),
|
||||||
|
REM 登录态会缓存在该目录,首次请手动登录 Pinterest。
|
||||||
|
|
||||||
|
setlocal
|
||||||
|
set SCRIPT_DIR=%~dp0
|
||||||
|
set UD=%SCRIPT_DIR%.chrome_session\User Data
|
||||||
|
if not exist "%UD%" mkdir "%UD%"
|
||||||
|
|
||||||
|
set CHROME="C:\Program Files\Google\Chrome\Application\chrome.exe"
|
||||||
|
if not exist %CHROME% set CHROME="%LOCALAPPDATA%\Google\Chrome\Application\chrome.exe"
|
||||||
|
|
||||||
|
start "" %CHROME% ^
|
||||||
|
--user-data-dir="%UD%" ^
|
||||||
|
--remote-debugging-port=9222 ^
|
||||||
|
--no-first-run ^
|
||||||
|
--no-default-browser-check ^
|
||||||
|
--disable-blink-features=AutomationControlled
|
||||||
|
|
||||||
|
echo 已打开调试版 Chrome(监听 9222),保持窗口打开。
|
||||||
|
echo 若 Pinterest 需要登录,请在此窗口手动登录一次(登录态会缓存在 .chrome_session)。
|
||||||
|
echo 之后运行:uv run pinterest_image_capture.py "关键词" -n 50 --cdp http://127.0.0.1:9222
|
||||||
|
pause
|
||||||
|
endlocal
|
||||||
@@ -0,0 +1,408 @@
|
|||||||
|
import argparse
|
||||||
|
import asyncio
|
||||||
|
import mimetypes
|
||||||
|
import os
|
||||||
|
import random
|
||||||
|
import socket
|
||||||
|
import urllib.request
|
||||||
|
from urllib.parse import urlparse, quote
|
||||||
|
|
||||||
|
import aiohttp
|
||||||
|
from aiofiles import open as aioopen
|
||||||
|
from playwright.async_api import Page, async_playwright
|
||||||
|
|
||||||
|
|
||||||
|
def get_filename_from_url(url: str, idx: int) -> str:
|
||||||
|
parsed = urlparse(url)
|
||||||
|
name = os.path.basename(parsed.path)
|
||||||
|
if not name: # 有些URL没有文件名
|
||||||
|
name = f"img_{idx}"
|
||||||
|
# 如果没有扩展名,尝试补上
|
||||||
|
if not os.path.splitext(name)[1]:
|
||||||
|
ext = mimetypes.guess_extension(parsed.path.split("?")[0])
|
||||||
|
name += ext if ext else ".jpg"
|
||||||
|
return name
|
||||||
|
|
||||||
|
|
||||||
|
def _probe_proxy(url: str) -> bool:
|
||||||
|
"""实测该地址是否是一个可用的 HTTP/HTTPS 代理(短超时 HEAD 探测)"""
|
||||||
|
proxy_handler = urllib.request.ProxyHandler({"http": url, "https": url})
|
||||||
|
opener = urllib.request.build_opener(proxy_handler)
|
||||||
|
try:
|
||||||
|
req = urllib.request.Request(
|
||||||
|
"http://www.gstatic.com/generate_204", method="HEAD"
|
||||||
|
)
|
||||||
|
opener.open(req, timeout=2)
|
||||||
|
return True
|
||||||
|
except Exception:
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def _read_windows_registry_proxy() -> str | None:
|
||||||
|
"""直接读取 Windows 系统代理配置(设置 → 网络 → 代理),免去端口扫描猜测。
|
||||||
|
|
||||||
|
来源:注册表 HKCU\\Software\\Microsoft\\Windows\\CurrentVersion\\Internet Settings
|
||||||
|
- ProxyEnable == 1 时,ProxyServer 形如 "127.0.0.1:6696" 或 "http=127.0.0.1:8899;https=..."。
|
||||||
|
非 Windows 平台或读取失败返回 None。
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
import winreg # 仅 Windows 可用
|
||||||
|
except ImportError:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
with winreg.OpenKey(
|
||||||
|
winreg.HKEY_CURRENT_USER,
|
||||||
|
r"Software\Microsoft\Windows\CurrentVersion\Internet Settings",
|
||||||
|
) as key:
|
||||||
|
enabled, _ = winreg.QueryValueEx(key, "ProxyEnable")
|
||||||
|
if not enabled:
|
||||||
|
return None
|
||||||
|
proxy_server, _ = winreg.QueryValueEx(key, "ProxyServer")
|
||||||
|
if not proxy_server:
|
||||||
|
return None
|
||||||
|
# 可能是 "http=127.0.0.1:8899;https=127.0.0.1:8899" 多协议格式,取第一个地址
|
||||||
|
first = proxy_server.split(";")[0]
|
||||||
|
if "=" in first:
|
||||||
|
first = first.split("=", 1)[1]
|
||||||
|
return ("http://" + first) if not first.startswith("http") else first
|
||||||
|
except Exception:
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def detect_proxy() -> str | None:
|
||||||
|
"""自动确定本地代理,无需手动填写。
|
||||||
|
|
||||||
|
优先级(确定性强的方式在前,端口扫描兜底在后):
|
||||||
|
1. 系统/用户环境变量 (HTTP_PROXY / HTTPS_PROXY ...)
|
||||||
|
2. 直接读 Windows 注册表系统代理 (Internet Settings / ProxyServer) —— 你配的 6696 直接命中
|
||||||
|
3. urllib 系统代理 (getproxies,已含注册表/环境变量)
|
||||||
|
4. 端口预检 + 实测本地常见代理端口(Clash / v2rayN / Shadowsocks 等)—— 仅作兜底
|
||||||
|
探测不到返回 None(调用方将直连)。
|
||||||
|
"""
|
||||||
|
# 1. 环境变量(系统代理 / CI 设置)
|
||||||
|
for env in ("HTTPS_PROXY", "https_proxy", "HTTP_PROXY", "http_proxy"):
|
||||||
|
val = os.environ.get(env)
|
||||||
|
if val:
|
||||||
|
return val.rstrip("/")
|
||||||
|
|
||||||
|
# 2. 直接读 Windows 注册表里的系统代理(你手动在系统设置里填的端口,这里精确拿到)
|
||||||
|
reg_proxy = _read_windows_registry_proxy()
|
||||||
|
if reg_proxy:
|
||||||
|
print(f"使用代理: {reg_proxy}(来自系统设置/注册表)")
|
||||||
|
return reg_proxy
|
||||||
|
|
||||||
|
# 3. urllib 系统代理兜底(已涵盖注册表/环境变量,跨平台)
|
||||||
|
sys_proxy = get_system_proxy()
|
||||||
|
if sys_proxy:
|
||||||
|
print(f"使用代理: {sys_proxy}(系统代理)")
|
||||||
|
return sys_proxy
|
||||||
|
|
||||||
|
# 4. 兜底:实测本地常见代理端口(先快速判断端口是否监听,避免无谓阻塞)
|
||||||
|
candidates = [
|
||||||
|
# Clash / Clash Verge / Clash for Windows
|
||||||
|
"127.0.0.1:7890", "127.0.0.1:7891", "127.0.0.1:7892", "127.0.0.1:7893",
|
||||||
|
"127.0.0.1:7894", "127.0.0.1:7878", "127.0.0.1:9090", # 9090 为 Clash 外部控制(也可能作代理)
|
||||||
|
# v2rayN / v2ray-core
|
||||||
|
"127.0.0.1:10808", "127.0.0.1:10809", "127.0.0.1:10810", "127.0.0.1:10811",
|
||||||
|
"127.0.0.1:10812", "127.0.0.1:10813", "127.0.0.1:10814", "127.0.0.1:10815",
|
||||||
|
# Shadowsocks / SSWindows
|
||||||
|
"127.0.0.1:1080", "127.0.0.1:1081", "127.0.0.1:1082", "127.0.0.1:1087",
|
||||||
|
"127.0.0.1:8388", "127.0.0.1:8389",
|
||||||
|
# Surge (mac/iOS 风格,本地也可能开)
|
||||||
|
"127.0.0.1:6152", "127.0.0.1:6153",
|
||||||
|
# Quantumult / Quantumult X
|
||||||
|
"127.0.0.1:6155", "127.0.0.1:6170",
|
||||||
|
# 系统代理 / HTTP 调试代理
|
||||||
|
"127.0.0.1:8888", "127.0.0.1:8080", "127.0.0.1:8081", "127.0.0.1:8088",
|
||||||
|
"127.0.0.1:3128", # 传统 squid 代理
|
||||||
|
# 其他常见:trojan / Brook / Netch / 蓝灯 / 自由门 / Proxifier
|
||||||
|
"127.0.0.1:10801", "127.0.0.1:10802", "127.0.0.1:10803", "127.0.0.1:10806",
|
||||||
|
"127.0.0.1:8118", # Privoxy
|
||||||
|
"127.0.0.1:10819", "127.0.0.1:2080", "127.0.0.1:1080",
|
||||||
|
]
|
||||||
|
for hp in candidates:
|
||||||
|
host, port = hp.split(":")
|
||||||
|
try:
|
||||||
|
with socket.create_connection((host, int(port)), timeout=0.4):
|
||||||
|
pass
|
||||||
|
except OSError:
|
||||||
|
continue # 端口没开,直接跳过(快速)
|
||||||
|
if _probe_proxy(f"http://{hp}"):
|
||||||
|
return f"http://{hp}"
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def get_system_proxy() -> str | None:
|
||||||
|
"""读取操作系统(Windows)设置的代理,作为探测不到本地代理时的兜底。
|
||||||
|
|
||||||
|
Playwright 启动的 Chrome 默认会继承系统代理;但 aiohttp 下载不会,
|
||||||
|
因此需要显式取出并传给下载阶段。
|
||||||
|
"""
|
||||||
|
proxies = urllib.request.getproxies()
|
||||||
|
val = proxies.get("https") or proxies.get("http")
|
||||||
|
return val.rstrip("/") if val else None
|
||||||
|
|
||||||
|
|
||||||
|
def _session_user_data_dir() -> str:
|
||||||
|
"""返回项目内持久化的 Chrome 用户数据目录。
|
||||||
|
|
||||||
|
该目录会一直保留(已加入 .gitignore),首次手动登录 Pinterest 后,
|
||||||
|
登录态(cookies 等)自动缓存在这里,后续运行直接复用,不再需要登录。
|
||||||
|
"""
|
||||||
|
base = os.path.dirname(os.path.abspath(__file__))
|
||||||
|
ud = os.path.join(base, ".chrome_session", "User Data")
|
||||||
|
os.makedirs(ud, exist_ok=True)
|
||||||
|
return ud
|
||||||
|
|
||||||
|
|
||||||
|
async def download_image(session: aiohttp.ClientSession, url: str, idx: int,
|
||||||
|
sem: asyncio.Semaphore, save_dir: str, proxy: str | None):
|
||||||
|
filename = get_filename_from_url(url, idx)
|
||||||
|
filepath = os.path.join(save_dir, filename)
|
||||||
|
|
||||||
|
async with sem: # 限制并发数
|
||||||
|
try:
|
||||||
|
async with session.get(url, proxy=proxy) as resp:
|
||||||
|
if resp.status == 200:
|
||||||
|
async with aioopen(filepath, "wb") as f:
|
||||||
|
await f.write(await resp.read())
|
||||||
|
print(f"✅ 下载成功: {filepath}")
|
||||||
|
else:
|
||||||
|
print(f"❌ 下载失败 {url} 状态码: {resp.status}")
|
||||||
|
except Exception as e:
|
||||||
|
print(f"⚠️ 下载错误 {url}: {e}")
|
||||||
|
|
||||||
|
|
||||||
|
async def download_all(imgs_url: set[str], save_dir: str, proxy: str | None):
|
||||||
|
sem = asyncio.Semaphore(10) # 同时最多10个下载任务
|
||||||
|
async with aiohttp.ClientSession() as session:
|
||||||
|
tasks = [
|
||||||
|
download_image(session, url, idx, sem, save_dir, proxy)
|
||||||
|
for idx, url in enumerate(imgs_url)
|
||||||
|
]
|
||||||
|
await asyncio.gather(*tasks)
|
||||||
|
|
||||||
|
|
||||||
|
async def human_move(page: Page, target_x: int, target_y: int):
|
||||||
|
"""模拟人工鼠标移动: 分步插值 + 随机抖动, 轨迹带弧度"""
|
||||||
|
# 获取当前鼠标位置(自己维护, playwright 不提供查询)
|
||||||
|
cur_x, cur_y = getattr(human_move, "_pos", (random.randint(100, 800), random.randint(100, 500)))
|
||||||
|
steps = random.randint(15, 30)
|
||||||
|
# 随机控制点让轨迹带弧度(近似贝塞尔)
|
||||||
|
ctrl_x = (cur_x + target_x) / 2 + random.randint(-150, 150)
|
||||||
|
ctrl_y = (cur_y + target_y) / 2 + random.randint(-150, 150)
|
||||||
|
for i in range(1, steps + 1):
|
||||||
|
t = i / steps
|
||||||
|
# 二次贝塞尔插值
|
||||||
|
x = (1 - t) ** 2 * cur_x + 2 * (1 - t) * t * ctrl_x + t ** 2 * target_x + random.uniform(-2, 2)
|
||||||
|
y = (1 - t) ** 2 * cur_y + 2 * (1 - t) * t * ctrl_y + t ** 2 * target_y + random.uniform(-2, 2)
|
||||||
|
await page.mouse.move(x, y)
|
||||||
|
await page.wait_for_timeout(random.randint(5, 20)) # 毫秒级间隔, 模拟手部移动速度
|
||||||
|
human_move._pos = (target_x, target_y)
|
||||||
|
|
||||||
|
|
||||||
|
async def human_scroll(page: Page, distance: int):
|
||||||
|
"""模拟人工滚动: 把总距离拆成多次小幅滚轮事件, 逐段发出"""
|
||||||
|
# 先把鼠标移到页面内一个随机位置再滚
|
||||||
|
await human_move(page, random.randint(200, 1000), random.randint(200, 700))
|
||||||
|
remaining = distance
|
||||||
|
while remaining > 0:
|
||||||
|
step = min(random.randint(40, 120), remaining) # 一次滚轮约 40~120px
|
||||||
|
await page.mouse.wheel(0, step)
|
||||||
|
remaining -= step
|
||||||
|
await page.wait_for_timeout(random.randint(30, 100)) # 滚轮事件间隔
|
||||||
|
|
||||||
|
|
||||||
|
async def _ensure_logged_in(page: Page, search_url: str) -> bool:
|
||||||
|
"""独立登录检查方法:在已跳到搜索页的前提下确认 Pinterest 已登录。
|
||||||
|
|
||||||
|
判定策略(避免误判):
|
||||||
|
- 以「未登录标志」为准:页面上一旦出现 "Log in" / "Sign up" 按钮,
|
||||||
|
才认定未登录;否则默认已登录(不依赖可能不匹配的已登录选择器)。
|
||||||
|
- Pinterest 是 SPA,goto 后需等待渲染,否则瞬间误判未登录。
|
||||||
|
- 已登录 → 立即返回 True,走直路。
|
||||||
|
- 未登录(被弹回登录墙)→ 回退首页提示手动登录一次,登录成功后返回 True。
|
||||||
|
- 一直未登录(用户关窗口/放弃)→ 返回 False。
|
||||||
|
"""
|
||||||
|
async def _has_login_wall() -> bool:
|
||||||
|
"""检测是否存在未登录标志(Log in / Sign up 按钮)。"""
|
||||||
|
try:
|
||||||
|
if await page.get_by_role("button", name="Log in").count():
|
||||||
|
return True
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
try:
|
||||||
|
if await page.get_by_role("button", name="Sign up").count():
|
||||||
|
return True
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
# 兜底:URL 被重定向到 /login 也是未登录的强信号
|
||||||
|
try:
|
||||||
|
if "/login" in page.url:
|
||||||
|
return True
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
return False
|
||||||
|
|
||||||
|
# 已经在搜索页了。等待 SPA 渲染,避免刚加载就被误判。
|
||||||
|
# 给一点时间让导航/按钮渲染出来(最多等 8 秒,出现登录墙或超时即停)。
|
||||||
|
try:
|
||||||
|
await page.wait_for_load_state("networkidle", timeout=8000)
|
||||||
|
except Exception:
|
||||||
|
pass # 网络一直不 idle 也不要卡死,继续判断
|
||||||
|
|
||||||
|
# 已登录:页面上找不到登录墙 → 直接走直路
|
||||||
|
if not await _has_login_wall():
|
||||||
|
print("✅ 已检测到登录态,直接开始搜索")
|
||||||
|
return True
|
||||||
|
|
||||||
|
# 未登录(搜索页被弹回登录墙):回退首页,提示手动登录一次,成功后立即继续
|
||||||
|
print("⚠️ 当前未登录(搜索页被拦截)。请在弹出的浏览器窗口中手动登录,登录成功后将自动继续……")
|
||||||
|
try:
|
||||||
|
await page.goto("https://www.pinterest.com/", wait_until="domcontentloaded")
|
||||||
|
# 登录成功后登录墙消失(Log in/Sign up 按钮不再存在)即视为登录
|
||||||
|
await page.wait_for_function(
|
||||||
|
"""() => {
|
||||||
|
const btns = [...document.querySelectorAll('button')];
|
||||||
|
const hasLogin = btns.some(b => /log\\s*in/i.test(b.textContent || ''));
|
||||||
|
const hasSignup = btns.some(b => /sign\\s*up/i.test(b.textContent || ''));
|
||||||
|
return !hasLogin && !hasSignup;
|
||||||
|
}""",
|
||||||
|
timeout=0, # 0 = 一直等到出现为止,不超时
|
||||||
|
)
|
||||||
|
print("✅ 登录成功,继续搜索")
|
||||||
|
return True
|
||||||
|
except Exception:
|
||||||
|
# 用户关掉页面 / 主动放弃
|
||||||
|
print("❌ 未检测到登录(页面已关闭或放弃登录)。请先登录后再运行脚本。")
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
async def scrape(keyword: str, count: int, headless: bool = False,
|
||||||
|
proxy: str | None = None,
|
||||||
|
cdp_url: str | None = None) -> set[str]:
|
||||||
|
imgs_url: set[str] = set()
|
||||||
|
async with async_playwright() as p:
|
||||||
|
if cdp_url:
|
||||||
|
# 高级模式:直接连用户已开好的调试版主浏览器(如 open_chrome_debug.bat)
|
||||||
|
browser = await p.chromium.connect_over_cdp(cdp_url)
|
||||||
|
context = browser.contexts[0] if browser.contexts else await browser.new_context()
|
||||||
|
print(f"已接管本机调试 Chrome: {cdp_url}")
|
||||||
|
else:
|
||||||
|
# 默认模式:用项目内持久化的 Chrome 用户数据目录启动一个可见窗口。
|
||||||
|
# 首次运行请在弹出的窗口里登录 Pinterest;登录态会缓存在
|
||||||
|
# .chrome_session 目录中,之后运行自动复用,无需再次登录。
|
||||||
|
# 复用系统已装的 Chrome(channel="chrome"),无需下载浏览器内核。
|
||||||
|
# 注意:Playwright 设置用户数据目录必须用 launch_persistent_context,
|
||||||
|
# 它返回的是 context(而非 browser),且该 context 已带登录态。
|
||||||
|
launch_kwargs = {
|
||||||
|
"headless": headless,
|
||||||
|
"channel": "chrome",
|
||||||
|
"user_data_dir": _session_user_data_dir(),
|
||||||
|
"args": [
|
||||||
|
"--disable-blink-features=AutomationControlled",
|
||||||
|
"--no-first-run",
|
||||||
|
"--no-default-browser-check",
|
||||||
|
],
|
||||||
|
}
|
||||||
|
if proxy:
|
||||||
|
launch_kwargs["proxy"] = {"server": proxy}
|
||||||
|
context = await p.chromium.launch_persistent_context(**launch_kwargs)
|
||||||
|
browser = context.browser
|
||||||
|
print("已启动本地 Chrome 窗口(可见,登录态缓存在 .chrome_session)")
|
||||||
|
|
||||||
|
page: Page = await context.new_page()
|
||||||
|
|
||||||
|
# 先直奔搜索页(有缓存/已登录时一条直路)
|
||||||
|
search_url = f"https://www.pinterest.com/search/pins/?q={quote(keyword)}"
|
||||||
|
await page.goto(search_url, wait_until="domcontentloaded")
|
||||||
|
|
||||||
|
# 独立登录检查:已登录直接开始;未登录才回退首页等手动登录
|
||||||
|
logged_in = await _ensure_logged_in(page, search_url)
|
||||||
|
if not logged_in:
|
||||||
|
# 关闭浏览器(持久化目录已保存任何已有状态),中止本次爬取
|
||||||
|
if cdp_url:
|
||||||
|
await browser.close()
|
||||||
|
else:
|
||||||
|
await context.close()
|
||||||
|
return set()
|
||||||
|
|
||||||
|
no_new_rounds = 0
|
||||||
|
while len(imgs_url) < count and no_new_rounds < 10:
|
||||||
|
await page.locator('div[role="listitem"]').first.wait_for(state="attached", timeout=50_000)
|
||||||
|
before = len(imgs_url)
|
||||||
|
imgs = await page.locator('div[role="listitem"]').all()
|
||||||
|
print(f"已加载 {len(imgs)} 个元素, 已收集 {len(imgs_url)} 张图片")
|
||||||
|
for img in imgs:
|
||||||
|
if len(imgs_url) >= count:
|
||||||
|
break
|
||||||
|
el = img.locator("img").first
|
||||||
|
if not await el.count():
|
||||||
|
continue
|
||||||
|
|
||||||
|
srcset = await el.get_attribute("srcset")
|
||||||
|
if not srcset:
|
||||||
|
continue
|
||||||
|
|
||||||
|
candidates = [
|
||||||
|
(s.split()[0], float(s.split()[1][:-1])) # (url, 倍率)
|
||||||
|
for s in srcset.split(",")
|
||||||
|
]
|
||||||
|
max_url = max(candidates, key=lambda x: x[1])[0]
|
||||||
|
imgs_url.add(max_url)
|
||||||
|
|
||||||
|
# 模拟人工: 鼠标平滑移动到随机位置 + 分段连续滚动,
|
||||||
|
# 平时停 1.5~3 秒, 偶尔长停顿(像在看图)
|
||||||
|
await human_scroll(page, random.randint(600, 1400))
|
||||||
|
if random.random() < 0.2:
|
||||||
|
await page.wait_for_timeout(random.randint(3000, 6000))
|
||||||
|
else:
|
||||||
|
await page.wait_for_timeout(random.randint(1500, 3000))
|
||||||
|
no_new_rounds = no_new_rounds + 1 if len(imgs_url) == before else 0
|
||||||
|
|
||||||
|
# 关闭浏览器(持久化目录中的登录态已自动保存,下次运行直接复用)
|
||||||
|
if cdp_url:
|
||||||
|
await browser.close() # CDP 接管模式:只断开,不关主浏览器
|
||||||
|
else:
|
||||||
|
await context.close() # 持久化 context 模式:关闭即保存登录态
|
||||||
|
|
||||||
|
return imgs_url
|
||||||
|
|
||||||
|
|
||||||
|
async def main():
|
||||||
|
parser = argparse.ArgumentParser(description="Pinterest 关键词图片爬取工具 (本地 Chrome + 持久化登录缓存)")
|
||||||
|
parser.add_argument("keyword", help="搜索关键词")
|
||||||
|
parser.add_argument("-n", "--count", type=int, default=40, help="爬取图片数量 (默认 40)")
|
||||||
|
parser.add_argument("--proxy", default=None,
|
||||||
|
help="下载/浏览使用的代理 (默认自动探测本地端口; 传空字符串禁用)")
|
||||||
|
parser.add_argument("--cdp", default=None,
|
||||||
|
help="可选:指定 CDP 地址以接管你已用调试模式打开的主浏览器 "
|
||||||
|
"(见 open_chrome_debug.bat)。不填则默认用项目内持久化缓存目录启动 Chrome,"
|
||||||
|
"首次登录后自动缓存登录态")
|
||||||
|
parser.add_argument("--headless", action="store_true",
|
||||||
|
help="自动启动 Chrome 时采用无头模式 (默认显示浏览器窗口)")
|
||||||
|
parser.add_argument("-o", "--output", default=None, help="保存目录 (默认 output/img/<关键词>)")
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
# 代理解析(命令行 > 本地端口探测 > 系统代理 > 直连);空字符串明确禁用
|
||||||
|
proxy = args.proxy
|
||||||
|
if proxy is None:
|
||||||
|
proxy = detect_proxy() or get_system_proxy()
|
||||||
|
if proxy:
|
||||||
|
print(f"使用代理: {proxy}(本地探测/系统代理)")
|
||||||
|
else:
|
||||||
|
print("未检测到代理, 将直连 (如无法访问 Pinterest 请手动指定 --proxy)")
|
||||||
|
proxy = proxy or None # 空字符串 -> 禁用
|
||||||
|
|
||||||
|
save_dir = args.output or os.path.join(os.getcwd(), "output", "img", args.keyword)
|
||||||
|
os.makedirs(save_dir, exist_ok=True)
|
||||||
|
print(f"关键词: {args.keyword} | 数量: {args.count} | 模式: {'接管主浏览器' if args.cdp else '本地窗口(缓存登录态)'} | 代理: {proxy or '无'}")
|
||||||
|
print(f"保存目录: {save_dir}")
|
||||||
|
|
||||||
|
imgs_url = await scrape(args.keyword, args.count, args.headless, proxy, args.cdp)
|
||||||
|
await download_all(imgs_url, save_dir, proxy)
|
||||||
|
print(f"完成, 共收集 {len(imgs_url)} 张图片")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
asyncio.run(main())
|
||||||
@@ -0,0 +1,13 @@
|
|||||||
|
[project]
|
||||||
|
name = "pinterest-scraper"
|
||||||
|
version = "0.1.0"
|
||||||
|
description = "Pinterest 关键词图片爬取工具,通过 CDP 连接本地浏览器"
|
||||||
|
requires-python = ">=3.10"
|
||||||
|
dependencies = [
|
||||||
|
"playwright>=1.40",
|
||||||
|
"aiohttp>=3.9",
|
||||||
|
"aiofiles>=23.0",
|
||||||
|
]
|
||||||
|
|
||||||
|
[tool.uv]
|
||||||
|
package = false
|
||||||
@@ -0,0 +1,50 @@
|
|||||||
|
"""Pinterest 爬取封装:把 pinterest_image_capture 的 CLI 逻辑封装成可调用函数,供 UI/节点后台线程调用。
|
||||||
|
|
||||||
|
用法:
|
||||||
|
from pinterest_scraper.scraper import scrape_pinterest
|
||||||
|
files = scrape_pinterest("vintage 70s", count=40, save_dir="output/pinterest_ref/US/vintage_70s")
|
||||||
|
"""
|
||||||
|
import asyncio
|
||||||
|
import os
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import List, Optional
|
||||||
|
|
||||||
|
from pinterest_scraper.pinterest_image_capture import (
|
||||||
|
detect_proxy,
|
||||||
|
download_all,
|
||||||
|
get_system_proxy,
|
||||||
|
scrape,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def scrape_pinterest(keyword: str, count: int = 40, save_dir: Optional[str] = None,
|
||||||
|
proxy: Optional[str] = None, headless: bool = False,
|
||||||
|
cdp_url: Optional[str] = None) -> List[str]:
|
||||||
|
"""按关键词爬取 Pinterest 图片并下载到 save_dir,返回下载成功的文件路径列表。
|
||||||
|
|
||||||
|
- proxy 为 None 时自动探测(环境变量/系统代理/本地常见端口);空字符串显式禁用。
|
||||||
|
- 首次运行会弹出本地 Chrome 窗口,需手动登录 Pinterest 一次,登录态缓存在
|
||||||
|
pinterest_scraper/.chrome_session,之后自动复用。
|
||||||
|
"""
|
||||||
|
# 代理解析:显式传入 > 自动探测 > 直连
|
||||||
|
if proxy is None:
|
||||||
|
proxy = detect_proxy() or get_system_proxy()
|
||||||
|
if proxy:
|
||||||
|
print(f"[pinterest] 使用代理: {proxy}")
|
||||||
|
else:
|
||||||
|
print("[pinterest] 未检测到代理,将直连(如无法访问请手动指定代理)")
|
||||||
|
proxy = proxy or None
|
||||||
|
|
||||||
|
out = save_dir or os.path.join(os.getcwd(), "output", "img", keyword)
|
||||||
|
os.makedirs(out, exist_ok=True)
|
||||||
|
print(f"[pinterest] 关键词: {keyword} | 数量: {count} | 保存目录: {out}")
|
||||||
|
|
||||||
|
imgs_url = asyncio.run(scrape(keyword, count, headless, proxy, cdp_url))
|
||||||
|
if not imgs_url:
|
||||||
|
print(f"[pinterest] 未收集到图片(可能未登录或搜索无结果): {keyword}")
|
||||||
|
return []
|
||||||
|
asyncio.run(download_all(imgs_url, out, proxy))
|
||||||
|
|
||||||
|
files = [str(p) for p in sorted(Path(out).iterdir()) if p.is_file()]
|
||||||
|
print(f"[pinterest] 完成,共下载 {len(files)} 张图片: {keyword}")
|
||||||
|
return files
|
||||||
@@ -190,8 +190,12 @@ def _apply_provider_mode(config: dict, provider: str) -> None:
|
|||||||
cp["backend"] = "openai" if on else "mock"
|
cp["backend"] = "openai" if on else "mock"
|
||||||
pp = config.setdefault("product", {})
|
pp = config.setdefault("product", {})
|
||||||
pp["backend"] = "openai" if on else "mock"
|
pp["backend"] = "openai" if on else "mock"
|
||||||
|
# Pinterest 参考模式:LLM(搜索词/图片分析)提供商跟随模式开关
|
||||||
|
pin = config.setdefault("pinterest", {})
|
||||||
|
pin["provider"] = "openai" if on else "mock"
|
||||||
print(f"[UI] 模式开关: {'OpenAI(真 LLM + 真生图)' if on else 'Mock(演示)'} | "
|
print(f"[UI] 模式开关: {'OpenAI(真 LLM + 真生图)' if on else 'Mock(演示)'} | "
|
||||||
f"llm_screen={ls['provider']} compose={cp['backend']} product={pp['backend']}")
|
f"llm_screen={ls['provider']} compose={cp['backend']} product={pp['backend']} "
|
||||||
|
f"pinterest={pin['provider']}")
|
||||||
|
|
||||||
|
|
||||||
def fetch_keywords(country, provider, max_seeds, log_q, oai=None):
|
def fetch_keywords(country, provider, max_seeds, log_q, oai=None):
|
||||||
@@ -374,6 +378,75 @@ def run_pipeline(countries, provider, max_seeds, log_q,
|
|||||||
log_q.put(("log", f"[UI] 缓存打包失败: {e}\n"))
|
log_q.put(("log", f"[UI] 缓存打包失败: {e}\n"))
|
||||||
|
|
||||||
|
|
||||||
|
def run_pinterest_pipeline(countries, provider, log_q, spu_tasks=None, spu_count=0,
|
||||||
|
oai=None, markup_percent=0.0, code_prefix="DG", template_path=""):
|
||||||
|
"""Pinterest 参考模式后台线程:独立于 Google Trends 采集链路。
|
||||||
|
|
||||||
|
各国独立种子词池 → LLM 搜索词(json_schema + 动态注入防重复)→ 爬图
|
||||||
|
→ LLM 分析图片 → 设计简报 → 产品生成(设计稿/主图/种草图/模板导出)。
|
||||||
|
"""
|
||||||
|
config = load_config()
|
||||||
|
config["seed_provider"] = provider
|
||||||
|
apply_openai_cfg(config, oai)
|
||||||
|
_apply_provider_mode(config, provider) # 联动 pinterest.provider / compose / product backend
|
||||||
|
p = config.setdefault("product", {})
|
||||||
|
if template_path:
|
||||||
|
p["template_path"] = template_path
|
||||||
|
if spu_tasks:
|
||||||
|
p["spu_tasks"] = spu_tasks
|
||||||
|
p.pop("spu_code", None)
|
||||||
|
if spu_count:
|
||||||
|
p["spu_count"] = spu_count
|
||||||
|
if markup_percent:
|
||||||
|
p["markup_percent"] = markup_percent
|
||||||
|
if code_prefix:
|
||||||
|
p["code_prefix"] = code_prefix
|
||||||
|
# 任务扩展(与 run_pipeline 一致):每个集合按自己数量复制 N 份
|
||||||
|
if spu_tasks:
|
||||||
|
tasks_list = []
|
||||||
|
for t in spu_tasks:
|
||||||
|
n = int(t.get("count") or 0) or spu_count or 1
|
||||||
|
for _ in range(n):
|
||||||
|
tt = dict(t)
|
||||||
|
tt.pop("count", None)
|
||||||
|
tasks_list.append(tt)
|
||||||
|
p["spu_tasks"] = tasks_list
|
||||||
|
if spu_count:
|
||||||
|
p["spu_count"] = spu_count
|
||||||
|
# 简报数上限联动:设计数 ≥ 任务数(否则任务会复用简报)
|
||||||
|
pcfg = config.setdefault("pinterest", {})
|
||||||
|
pcfg["max_designs"] = max(int(pcfg.get("max_designs", 10) or 10), int(spu_count or 1))
|
||||||
|
|
||||||
|
old_stdout = sys.stdout
|
||||||
|
sys.stdout = StdoutRedirector(log_q)
|
||||||
|
results = {}
|
||||||
|
task_ts = time.strftime("%Y%m%d%H%M%S")
|
||||||
|
try:
|
||||||
|
from graph.agent import run_pinterest_ref
|
||||||
|
for c in countries:
|
||||||
|
log_q.put(("log", f"\n===== Pinterest 参考模式 {c}(SPU 数量 {spu_count})=====\n"))
|
||||||
|
state = run_pinterest_ref(c, config, config_root(), runtime_root(), task_timestamp=task_ts)
|
||||||
|
items = state.get("product") or []
|
||||||
|
errs = state.get("errors") or []
|
||||||
|
log_q.put(("log", f"[{c}] Pinterest 参考完成:{len(items)} 个产品,兜底错误 {len(errs)}\n"))
|
||||||
|
results[c] = items
|
||||||
|
ts_dir = runtime_root() / "output" / countries[0] / task_ts
|
||||||
|
log_q.put(("log", f"\n✅ Pinterest 参考任务完成,产物文件夹:{ts_dir}\n"
|
||||||
|
f" (缓存/去重记录在 {runtime_root() / 'output' / countries[0]} 根目录,不进任务文件夹)\n"))
|
||||||
|
log_q.put(("done", results))
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
import traceback
|
||||||
|
log_q.put(("log", f"运行失败: {e}\n{traceback.format_exc()}\n"))
|
||||||
|
log_q.put(("error", None))
|
||||||
|
finally:
|
||||||
|
sys.stdout = old_stdout
|
||||||
|
try:
|
||||||
|
for c in countries:
|
||||||
|
_pack_cache(c, log_q) # 运行完成 → 自动打包缓存/去重数据
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
log_q.put(("log", f"[UI] 缓存打包失败: {e}\n"))
|
||||||
|
|
||||||
|
|
||||||
def _pack_cache(country: str, log_q=None) -> str:
|
def _pack_cache(country: str, log_q=None) -> str:
|
||||||
"""运行/采集完成后,把该国缓存+去重数据打包成 zip(output/cache_packs/<国>_<时间戳>.zip),
|
"""运行/采集完成后,把该国缓存+去重数据打包成 zip(output/cache_packs/<国>_<时间戳>.zip),
|
||||||
内含 design_briefs / used_designs / collected_keywords / products.json + .cache 关键缓存,
|
内含 design_briefs / used_designs / collected_keywords / products.json + .cache 关键缓存,
|
||||||
@@ -459,6 +532,12 @@ class App(tk.Tk):
|
|||||||
variable=self.provider_var).pack(side="left", padx=2)
|
variable=self.provider_var).pack(side="left", padx=2)
|
||||||
ttk.Radiobutton(top, text="Mock 演示", value="mock",
|
ttk.Radiobutton(top, text="Mock 演示", value="mock",
|
||||||
variable=self.provider_var).pack(side="left", padx=2)
|
variable=self.provider_var).pack(side="left", padx=2)
|
||||||
|
ttk.Label(top, text=" 流程:").pack(side="left", padx=(14, 0))
|
||||||
|
self.flow_var = tk.StringVar(value="trends")
|
||||||
|
ttk.Radiobutton(top, text="热点采集", value="trends",
|
||||||
|
variable=self.flow_var).pack(side="left", padx=2)
|
||||||
|
ttk.Radiobutton(top, text="Pinterest 参考", value="pinterest",
|
||||||
|
variable=self.flow_var).pack(side="left", padx=2)
|
||||||
ttk.Label(top, text=" 种子数量:").pack(side="left", padx=(14, 0))
|
ttk.Label(top, text=" 种子数量:").pack(side="left", padx=(14, 0))
|
||||||
self.seed_var = tk.StringVar(value="24")
|
self.seed_var = tk.StringVar(value="24")
|
||||||
ttk.Entry(top, textvariable=self.seed_var, width=4).pack(side="left")
|
ttk.Entry(top, textvariable=self.seed_var, width=4).pack(side="left")
|
||||||
@@ -934,6 +1013,17 @@ class App(tk.Tk):
|
|||||||
self._busy = True
|
self._busy = True
|
||||||
self.run_btn.config(state="disabled", text="运行中…")
|
self.run_btn.config(state="disabled", text="运行中…")
|
||||||
self.fetch_btn.config(state="disabled")
|
self.fetch_btn.config(state="disabled")
|
||||||
|
if self.flow_var.get() == "pinterest":
|
||||||
|
# Pinterest 参考模式:独立于 Google Trends 采集链路
|
||||||
|
threading.Thread(
|
||||||
|
target=run_pinterest_pipeline,
|
||||||
|
args=(countries, self.provider_var.get(), self._q,
|
||||||
|
tasks, spu_count, self._oai_cfg(), markup,
|
||||||
|
self.code_prefix_var.get().strip() or "DG",
|
||||||
|
self.template_path_var.get().strip()),
|
||||||
|
daemon=True,
|
||||||
|
).start()
|
||||||
|
else:
|
||||||
threading.Thread(
|
threading.Thread(
|
||||||
target=run_pipeline,
|
target=run_pipeline,
|
||||||
args=(countries, self.provider_var.get(), ms, self._q,
|
args=(countries, self.provider_var.get(), ms, self._q,
|
||||||
|
|||||||
Reference in New Issue
Block a user