Files
crawler-plugin/app/brand_spider/main.py
铭坤 bbca159684 modified: blueprints/__pycache__/__init__.cpython-39.pyc
modified:   blueprints/__pycache__/admin.cpython-39.pyc
	modified:   blueprints/__pycache__/auth.cpython-39.pyc
	modified:   blueprints/__pycache__/brand.cpython-311.pyc
	modified:   blueprints/__pycache__/brand.cpython-39.pyc
	modified:   blueprints/brand.py
	new file:   brand_spider/__pycache__/main.cpython-311.pyc
	modified:   brand_spider/__pycache__/main.cpython-39.pyc
	new file:   brand_spider/__pycache__/web_dec.cpython-311.pyc
	modified:   brand_spider/main.py
	modified:   config.py
	modified:   web_source/admin.html
	modified:   web_source/brand.html
	modified:   web_source/home.html
	modified:   web_source/index.html
	modified:   web_source/login.html
	new file:   web_source/templates_backup/admin.html
	new file:   web_source/templates_backup/brand.html
	new file:   web_source/templates_backup/home.html
	new file:   web_source/templates_backup/index.html
	new file:   web_source/templates_backup/login.html
2026-03-26 16:39:39 +08:00

605 lines
19 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import json
import re
import random
import requests
import time
import unicodedata
from brand_spider.web_dec import decrypt_via_service, get_guid
try:
from config import proxy_url as CONFIG_PROXY_URL, proxy_mode as CONFIG_PROXY_MODE
except ImportError:
CONFIG_PROXY_URL = None
CONFIG_PROXY_MODE = 1
# Forbidden 后 3 分钟内统一使用代理:记录代理生效截止时间与当前代理
_FORBIDDEN_PROXY_UNTIL = 0.0
_FORBIDDEN_PROXY_DICT = None
_FORBIDDEN_PROXY_MINUTES = 0.5
special_char_pattern = re.compile(r'[^\w\s]')
def remove_accents(input_str):
# 将字符分解为基础字符和重音符号
nksel = unicodedata.normalize('NFKD', input_str)
# 过滤掉非间距重音符号,并重新编码
return "".join([c for c in nksel if not unicodedata.combining(c)])
def clean_text(text):
text = remove_accents(text)
return special_char_pattern.sub(' ', text).replace(" ","")
def create_code_generator():
counter = random.randint(0, 0xFFFFFF)
def get_code():
nonlocal counter
counter = (counter + 1) % 0xFFFFFF
low_16_bits = counter & 0xFFFF
return f"{low_16_bits:04x}"
return get_code()
def search(hashsearch, data, proxies=None):
global _FORBIDDEN_PROXY_UNTIL, _FORBIDDEN_PROXY_DICT
url = "https://api.branddb.wipo.int/search"
headers = {
"accept": "application/json, text/plain, */*",
"accept-language": "zh-CN,zh;q=0.9,en;q=0.8",
"cache-control": "no-cache",
"content-type": "application/json",
"hashsearch": hashsearch,
"origin": "https://branddb.wipo.int",
"pragma": "no-cache",
"priority": "u=1, i",
"referer": "https://branddb.wipo.int/",
"user-agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/145.0.0.0 Safari/537.36"
}
# 若 3 分钟内曾出现 Forbidden则直接使用当时保存的代理
now = time.time()
if now < _FORBIDDEN_PROXY_UNTIL and _FORBIDDEN_PROXY_DICT:
proxies = _FORBIDDEN_PROXY_DICT
print("处于 Forbidden 代理窗口内,直接使用代理",proxies)
# 发送 POST 请求
response = requests.post(url, headers=headers, json=data, proxies=proxies)
enc_text = response.text
# print("原始结果", enc_text)
# 若返回 Forbidden则从 config 的 proxy_url 获取代理 IP 后重试,并开启 3 分钟代理窗口
if (enc_text.strip() == '{"message":"Forbidden"}' or "Too Many Requests" in enc_text) and CONFIG_PROXY_URL:
try:
proxy_resp = requests.get(CONFIG_PROXY_URL, timeout=10)
print("代理请求结果->", proxy_resp.text)
# 模式 2账号密码代理接口返回 JSON
if CONFIG_PROXY_MODE == 2:
resp_json = proxy_resp.json()
proxy_list = resp_json.get("data", {}).get("list") or []
first_item = proxy_list[0] if proxy_list else None
if first_item:
ip = first_item.get("ip")
port = first_item.get("port")
account = first_item.get("account")
password = first_item.get("password")
if ip and port and account and password:
auth_proxy = f"{account}:{password}@{ip}:{port}"
print("获取到账号密码代理:", auth_proxy)
proxies = {
"http": f"http://{auth_proxy}",
"https": f"http://{auth_proxy}",
}
# 默认模式 1普通 IP:port 文本
if CONFIG_PROXY_MODE != 2:
proxy_ip = (proxy_resp.text or "").strip()
if proxy_ip:
proxies = {
"http": f"http://{proxy_ip}",
"https": f"http://{proxy_ip}",
}
if proxies:
response = requests.post(url, headers=headers, json=data, proxies=proxies)
enc_text = response.text
print("代理重试结果", enc_text)
# 记录 3 分钟内都使用该代理
_FORBIDDEN_PROXY_UNTIL = now + _FORBIDDEN_PROXY_MINUTES * 60
_FORBIDDEN_PROXY_DICT = proxies
except Exception as e:
print("获取代理或重试失败:", e)
return enc_text
class TaskCancelledError(Exception):
"""用户取消任务时抛出,供上层捕获并中止"""
pass
def single_file_handle(brand, strategy="Terms"):
status_info = {
"Ended": "已结束",
"Expired": "已过期的",
"Pending": "待决",
"Registered": "已注册",
"RegisteredMadrid": "国际注册有效",
"Unknown": "未知"
}
country_info = {
"AB": "ARABPAT",
"AD": "安道尔",
"AE": "UAE",
"AF": "阿富汗",
"AFR": "非洲",
"AG": "安提瓜和巴布达",
"AI": "安圭拉",
"AL": "阿尔巴尼亚",
"AM": "亚美尼亚",
"AN": "荷属安的列斯",
"ANT": "南极洲",
"AO": "安哥拉",
"AP": "非洲地区知识产权组织 ",
"AQ": "南极洲",
"AR": "阿根廷",
"AS": "美属萨摩亚",
"ASI": "亚洲",
"AT": "奥地利",
"AU": "澳大利亚",
"AW": "阿鲁巴",
"AX": "奥兰群岛",
"AZ": "阿塞拜疆",
"BA": "波斯尼亚和黑塞哥维纳",
"BB": "巴巴多斯",
"BD": "孟加拉国",
"BE": "比利时",
"BF": "布基纳法索",
"BG": "保加利亚",
"BH": "巴林",
"BI": "布隆迪",
"BJ": "贝宁",
"BL": "圣巴泰勒米",
"BM": "百慕大",
"BN": "文莱达鲁萨兰国",
"BO": "多民族玻利维亚国",
"BQ": "博纳尔、圣俄斯塔休斯和萨巴",
"BR": "巴西",
"BS": "巴哈马",
"BT": "不丹",
"BV": "布韦岛",
"BW": "博茨瓦纳",
"BX": "比荷卢知识产权局",
"BY": "白俄罗斯",
"BZ": "伯利兹",
"CA": "加拿大",
"CC": "科科斯群岛(基灵群岛)",
"CD": "刚果民主共和国",
"CF": "中非共和国",
"CG": "刚果",
"CH": "瑞士",
"CI": "科特迪瓦",
"CK": "库克群岛",
"CL": "智利",
"CM": "喀麦隆",
"CN": "中国",
"CO": "哥伦比亚",
"CR": "哥斯达黎加",
"CS": "捷克斯洛伐克",
"CU": "古巴",
"CV": "佛得角",
"CW": "库拉索",
"CX": "圣诞岛",
"CY": "塞浦路斯",
"CZ": "捷克共和国",
"DD": "德意志民主共和国",
"DE": "德国",
"DJ": "吉布提",
"DK": "丹麦",
"DM": "多米尼克",
"DO": "多米尼加",
"DT": "西德",
"DZ": "阿尔及利亚",
"EA": "欧亚专利组织",
"EC": "厄瓜多尔",
"EE": "爱沙尼亚",
"EG": "埃及",
"EH": "西撒哈拉",
"EM": "欧洲联盟",
"EP": "欧洲专利局",
"ER": "厄立特里亚",
"ES": "西班牙",
"ET": "埃塞俄比亚",
"EUR": "欧洲",
"FI": "芬兰",
"FJ": "斐济",
"FK": "福克兰群岛(马尔维纳斯群岛)",
"FM": "密克罗尼西亚联邦",
"FO": "法罗群岛",
"FR": "法国",
"GA": "加蓬",
"GB": "英国",
"GC": "海湾阿拉伯国家合作委员会专利局",
"GD": "格林纳达",
"GE": "格鲁吉亚",
"GF": "法属圭亚那",
"GG": "格恩西岛",
"GH": "加纳",
"GI": "直布罗陀",
"GL": "格陵兰",
"GM": "冈比亚",
"GN": "几内亚",
"GP": "瓜德罗普",
"GQ": "赤道几内亚",
"GR": "希腊",
"GS": "南乔治亚和南桑威奇群岛",
"GT": "危地马拉",
"GU": "关岛",
"GW": "几内亚比绍",
"GY": "圭亚那",
"HK": "香港",
"HM": "赫德岛和麦克唐纳群岛",
"HN": "洪都拉斯",
"HR": "克罗地亚",
"HT": "海地",
"HU": "匈牙利",
"IB": "世界知识产权组织国际局",
"ID": "印度尼西亚",
"IE": "爱尔兰",
"IL": "以色列",
"IM": "马恩岛",
"IN": "印度",
"INN": "世界卫生组织",
"IO": "英属印度洋领地",
"IQ": "伊拉克",
"IR": "伊朗伊斯兰共和国",
"IS": "冰岛",
"IT": "意大利",
"JE": "泽西岛",
"JM": "牙买加",
"JO": "约旦",
"JP": "日本",
"KE": "肯尼亚",
"KG": "吉尔吉斯斯坦",
"KH": "柬埔寨",
"KI": "基里巴斯",
"KM": "科摩罗",
"KN": "圣基茨和尼维斯",
"KP": "朝鲜民主主义人民共和国",
"KR": "大韩民国",
"KW": "科威特",
"KY": "开曼群岛",
"KZ": "哈萨克斯坦",
"LA": "老挝人民民主共和国",
"LB": "黎巴嫩",
"LC": "圣卢西亚",
"LI": "列支敦士登",
"LISBON": "WIPO",
"LK": "斯里兰卡",
"LP": "LATIPAT",
"LR": "利比里亚",
"LS": "莱索托",
"LT": "立陶宛",
"LU": "卢森堡",
"LV": "拉脱维亚",
"LY": "利比亚",
"MA": "摩洛哥",
"MC": "摩纳哥",
"MD": "摩尔多瓦共和国",
"ME": "黑山",
"MF": "圣马丁(法国部分)",
"MG": "马达加斯加",
"MH": "马绍尔群岛",
"MK": "北马其顿共和国",
"ML": "马里",
"MM": "缅甸",
"MN": "蒙古",
"MO": "澳门",
"MP": "北马里亚纳群岛",
"MQ": "马提尼克",
"MR": "毛里塔尼亚",
"MS": "蒙特塞拉特",
"MT": "马耳他",
"MU": "毛里求斯",
"MV": "马尔代夫",
"MW": "马拉维",
"MX": "墨西哥",
"MY": "马来西亚",
"MZ": "莫桑比克",
"NA": "纳米比亚",
"NAM": "北美洲",
"NC": "新喀里多尼亚",
"NE": "尼日尔",
"NF": "诺福克岛",
"NG": "尼日利亚",
"NI": "尼加拉瓜",
"NL": "荷兰",
"NO": "挪威",
"NP": "尼泊尔",
"NR": "瑙鲁",
"NU": "纽埃",
"NZ": "新西兰",
"OA": "非洲知识产权组织",
"OCE": "大洋洲",
"OM": "阿曼",
"PA": "巴拿马",
"PE": "秘鲁",
"PF": "法属波利尼西亚",
"PG": "巴布亚新几内亚",
"PH": "菲律宾",
"PK": "巴基斯坦",
"PL": "波兰",
"PM": "圣皮埃尔和密克隆",
"PN": "皮特凯恩",
"PR": "波多黎各",
"PS": "巴勒斯坦",
"PT": "葡萄牙",
"PW": "帕劳",
"PY": "巴拉圭",
"QA": "卡塔尔",
"QO": "没有ST.3代码的组织",
"QZ": "欧洲联盟",
"RE": "留尼汪",
"RO": "罗马尼亚",
"RS": "塞尔维亚",
"RU": "俄罗斯联邦",
"RW": "卢旺达",
"SA": "沙特阿拉伯",
"SAM": "南美洲",
"SB": "所罗门群岛",
"SC": "塞舌尔",
"SD": "苏丹",
"SE": "瑞典",
"SG": "新加坡",
"SH": "圣赫勒拿、阿森松和特里斯坦-达库尼亚",
"SI": "斯洛文尼亚",
"SIXTER": "WIPO",
"SJ": "斯瓦尔巴和扬马延",
"SK": "斯洛伐克",
"SL": "塞拉里昂",
"SM": "圣马力诺",
"SN": "塞内加尔",
"SO": "索马里",
"SR": "苏里南",
"SS": "南苏丹",
"ST": "圣多美和普林西比",
"SU": "苏联",
"SV": "萨尔瓦多",
"SX": "圣马丁(荷兰部分)",
"SY": "阿拉伯叙利亚共和国",
"SZ": "斯威士兰",
"TC": "特克斯和凯科斯群岛",
"TD": "乍得",
"TF": "法属南部领地",
"TG": "多哥",
"TH": "泰国",
"TJ": "塔吉克斯坦",
"TK": "托克劳",
"TL": "东帝汶",
"TM": "土库曼斯坦",
"TN": "突尼斯",
"TO": "汤加",
"TR": "土耳其",
"TT": "特立尼达和多巴哥",
"TV": "图瓦卢",
"TW": "台湾(中国的省)",
"TZ": "坦桑尼亚联合共和国",
"UA": "乌克兰",
"UG": "乌干达",
"UK": "UK",
"UM": "美国本土外小岛屿",
"US": "USA",
"UY": "乌拉圭",
"UZ": "乌兹别克斯坦",
"VA": "罗马教廷",
"VC": "圣文森特和格林纳丁斯",
"VE": "委内瑞拉玻利瓦尔共和国",
"VG": "英属维尔京群岛",
"VI": "美属维尔京群岛",
"VN": "越南",
"VU": "瓦努阿图",
"WF": "瓦利斯和富图纳",
"WHO": "世界卫生组织",
"WO": "WIPO",
"WS": "萨摩亚",
"XK": "科索沃共和国",
"XN": "北欧专利局",
"XX": "国际",
"XXX": "跨国和国际局",
"YD": "民主也门",
"YE": "也门",
"YT": "马约特岛",
"YU": "塞尔维亚和黑山",
"ZA": "南非",
"ZM": "赞比亚",
"ZW": "津巴布韦"
}
check_value = [
"瑞典",
"芬兰",
"丹麦",
"挪威",
"冰岛",
"法国",
"爱尔兰",
"荷兰",
"比利时",
"卢森堡",
"英国",
"摩纳哥",
"德国",
"波兰",
"捷克",
"斯洛伐克",
"匈牙利",
"奥地利",
"瑞士",
"列支敦士登",
"爱沙尼亚",
"拉脱维亚",
"立陶宛",
"俄罗斯",
"白俄罗斯",
"乌克兰",
"摩尔多瓦",
"西班牙",
"葡萄牙",
"意大利",
"希腊",
"斯洛文尼亚",
"克罗地亚",
"罗马尼亚",
"保加利亚",
"塞尔维亚",
"阿尔巴尼亚",
"黑山",
"马耳他",
"塞浦路斯",
"北马其顿",
"梵蒂冈",
"圣马力诺",
"安道尔",
"波黑",
"欧洲联盟"
]
check_staus = ["已过期的","已结束"]
# 5. 初始化不符合品牌的数据列表
faild_data = []
# 初始化查询失败的品牌的数据列表
query_faild_data = []
# Simple 策略下最多翻 3 页0,30,60直到命中 554 行判断
max_pages = 3 if strategy == "Simple" else 1
page = 0
matched = False
while page < max_pages and not matched:
try:
as_structure_dict = {
"_id": create_code_generator(),
"boolean": "AND",
"bricks": [
{
"_id": create_code_generator(),
"key": "brandName",
"value": brand,
"strategy": strategy
}
]
}
# 构造 asStructure 内部 JSON 对象
as_structure_dict = as_structure_dict
# print(as_structure_dict)
# 将内部对象转为 JSON 字符串(作为 asStructure 字段的值)
as_structure_str = json.dumps(as_structure_dict, ensure_ascii=False)
data = {
"sort": "score desc",
"rows": "30",
"asStructure": as_structure_str,
"fg": "_void_"
}
# 翻页:第二页 start=30第三页 start=60仅 Simple 策略)
if strategy == "Simple" and page > 0:
data["start"] = str(30 * page)
# print("请求参数", as_structure_dict, "页码:", page)
hashsearch = get_guid()
enc_text = search(hashsearch, data)
print(f"{hashsearch}】待解密-->", enc_text)
dec_text = decrypt_via_service(hashsearch, enc_text)
# print(type(dec_text))
dec_text = str(dec_text)
# print(f"解密之后的结果-->", dec_text)
data = json.loads(dec_text)
except Exception as e:
# 仅第一页请求失败时记录为查询失败品牌
if page == 0:
import traceback
traceback.print_exc()
safe_brand = brand.encode('gbk', errors='replace').decode('gbk')
print(f"品牌:{safe_brand},处理失败:{e}")
# query_faild_data.append({"brand": brand, "time": datetime.datetime.now().strftime("%y-%m-%d %H:%M:%S")})
query_faild_data.append(brand)
break
# print(data)
if data.get("response", {}).get("numFound", 0) > 0:
if strategy == "Simple":
new_brand_name = clean_text(brand)
_brand = new_brand_name.lower()
else:
_brand = brand
docs = data["response"]["docs"]
for d in docs:
office = d.get("office")
status = d.get("status")
registrationDate = d.get("registrationDate")
status_name = status_info.get(status, status)
brand_name = d.get("brandName")
if isinstance(brand_name, list):
brand_name = "|".join(brand_name)
if strategy == "Simple":
new_brand_name = clean_text(brand_name)
new_brand_name = new_brand_name.lower()
if _brand != new_brand_name:
continue
check_office = d.get("designation")
check_office.append(office)
for i in set(check_office):
office_name = country_info.get(i, i)
# print(office_name,office_name in check_value and status_name not in check_staus)
if office_name in check_value and status_name not in check_staus:
# 命中 554 行判断,记录并结束当前品牌后续翻页
# data_rows = [row for row in data_rows if row.get("品牌") != brand]
# print(office_name, office in ["EM","QZ"] and i not in ["EM","QZ"])
if office in ["EM","QZ"] and i not in ["EM","QZ"]:
continue
if strategy == "Simple":
faild_data.append({"brand": brand, "country": office_name, "status": status_name}) # "reason":"|".join(d.get("brandName","")
matched = True
else:
# print("添加",brand,office_name)
faild_data.append({"brand": brand, "country": office_name, "status": status_name})
# break
if matched:
break
# 如果 Simple 策略下本页未命中,则翻下一页;其他策略不翻页
if not matched:
page += 1
return faild_data,query_faild_data
if __name__ == '__main__':
file_path = r"D:\\私单交付\\maixiang_AI\\测试图片数据\\品牌采集测试\\brand_task_73\\源文件\\brand_task_75\\源文件\\3.10.xlsx"
res = single_file_handle(file_path,"结果.xlsx")
print(res)