import urllib.request import re import json req = urllib.request.Request('https://www.zhuig.com/community', headers={'User-Agent': 'Mozilla/5.0', 'Cache-Control': 'no-cache'}) html = urllib.request.urlopen(req, timeout=15).read().decode('utf-8') # 提取所有标签里的href和text: 板块名 # 但HTML结构可能嵌套。用BeautifulSoup风格的解析 # 找所有板块链接 pattern = re.compile(r']*href="/community/([a-zA-Z0-9_-]+)"[^>]*>(.*?)', re.DOTALL) matches = pattern.findall(html) print(f'找到 {len(matches)} 个链接') # 提取每个slug对应的板块名(去HTML标签) slug_to_names = {} for slug, text in matches: # 去掉HTML标签 clean = re.sub(r'<[^>]+>', '', text).strip() if clean and len(clean) < 20: slug_to_names.setdefault(slug, []).append(clean) print('\n=== 全部板块slug与显示名 ===') for slug in sorted(slug_to_names.keys()): names = slug_to_names[slug] print(f' {slug:30s} → {names}') # 对比数据库/API api_req = urllib.request.Request('https://www.zhuig.com/api/forum/categories', headers={'Cache-Control': 'no-cache'}) api_data = json.loads(urllib.request.urlopen(api_req, timeout=10).read().decode('utf-8')) api_slugs = [] for c in api_data['categories']: api_slugs.append((c['slug'], c['name'], '顶层')) for child in c.get('children', []): api_slugs.append((child['slug'], child['name'], ' ├─')) for gc in child.get('children', []): api_slugs.append((gc['slug'], gc['name'], ' └─')) print(f'\n=== API板块数: {len(api_slugs)} ===') # 看哪些是新版slug new_design_slugs = ['ec-platform', 'ec-livestream', 'ec-supply', 'ec-dtc', 'ec-private', 'ec-group', 'ai-tools-app', 'ai-llm', 'ai-startup-forum', 'ai-saas', 'ai-hardware', 'fd-restaurant', 'fd-tea', 'fd-prepared', 'fd-supply', 'fd-delivery', 're-residential', 're-commercial', 're-renovation', 're-property', 'fi-stock', 'fi-insurance', 'fi-pe', 'fi-forex', 'fi-crypto', 'ct-writing', 'ct-short-video', 'ct-live', 'ct-mcn', 'ct-podcast', 'll-beauty', 'll-fitness', 'll-pet', 'll-housekeeping', 'll-repair', 'll-edu-local', 'hl-cosmetic', 'hl-wellness', 'hl-elderly', 'hl-mental', 'hl-rehab', 'edu-knowledge', 'edu-skills', 'edu-abroad', 'edu-corporate', 'cb-ecommerce', 'cb-factory', 'cb-logistics', 'cb-payment', 'cb-brand'] api_slug_list = [s[0] for s in api_slugs] in_api = [s for s in new_design_slugs if s in api_slug_list] not_in_api = [s for s in new_design_slugs if s not in api_slug_list] print(f'\n新设计子板块: 总{len(new_design_slugs)}, API中有{len(in_api)}, 缺失{len(not_in_api)}') if not_in_api: print(f'缺失: {not_in_api}')