49 lines
2.6 KiB
Python
49 lines
2.6 KiB
Python
import urllib.request
|
|
import re
|
|
import json
|
|
|
|
req = urllib.request.Request('https://www.zhuig.com/community', headers={'User-Agent': 'Mozilla/5.0', 'Cache-Control': 'no-cache'})
|
|
html = urllib.request.urlopen(req, timeout=15).read().decode('utf-8')
|
|
|
|
# 提取所有<a>标签里的href和text: <a href="/community/xxx" ... >板块名</a>
|
|
# 但HTML结构可能嵌套。用BeautifulSoup风格的解析
|
|
# 找所有板块链接
|
|
pattern = re.compile(r'<a[^>]*href="/community/([a-zA-Z0-9_-]+)"[^>]*>(.*?)</a>', re.DOTALL)
|
|
matches = pattern.findall(html)
|
|
print(f'找到 {len(matches)} 个链接')
|
|
|
|
# 提取每个slug对应的板块名(去HTML标签)
|
|
slug_to_names = {}
|
|
for slug, text in matches:
|
|
# 去掉HTML标签
|
|
clean = re.sub(r'<[^>]+>', '', text).strip()
|
|
if clean and len(clean) < 20:
|
|
slug_to_names.setdefault(slug, []).append(clean)
|
|
|
|
print('\n=== 全部板块slug与显示名 ===')
|
|
for slug in sorted(slug_to_names.keys()):
|
|
names = slug_to_names[slug]
|
|
print(f' {slug:30s} → {names}')
|
|
|
|
# 对比数据库/API
|
|
api_req = urllib.request.Request('https://www.zhuig.com/api/forum/categories', headers={'Cache-Control': 'no-cache'})
|
|
api_data = json.loads(urllib.request.urlopen(api_req, timeout=10).read().decode('utf-8'))
|
|
api_slugs = []
|
|
for c in api_data['categories']:
|
|
api_slugs.append((c['slug'], c['name'], '顶层'))
|
|
for child in c.get('children', []):
|
|
api_slugs.append((child['slug'], child['name'], ' ├─'))
|
|
for gc in child.get('children', []):
|
|
api_slugs.append((gc['slug'], gc['name'], ' └─'))
|
|
|
|
print(f'\n=== API板块数: {len(api_slugs)} ===')
|
|
# 看哪些是新版slug
|
|
new_design_slugs = ['ec-platform', 'ec-livestream', 'ec-supply', 'ec-dtc', 'ec-private', 'ec-group', 'ai-tools-app', 'ai-llm', 'ai-startup-forum', 'ai-saas', 'ai-hardware', 'fd-restaurant', 'fd-tea', 'fd-prepared', 'fd-supply', 'fd-delivery', 're-residential', 're-commercial', 're-renovation', 're-property', 'fi-stock', 'fi-insurance', 'fi-pe', 'fi-forex', 'fi-crypto', 'ct-writing', 'ct-short-video', 'ct-live', 'ct-mcn', 'ct-podcast', 'll-beauty', 'll-fitness', 'll-pet', 'll-housekeeping', 'll-repair', 'll-edu-local', 'hl-cosmetic', 'hl-wellness', 'hl-elderly', 'hl-mental', 'hl-rehab', 'edu-knowledge', 'edu-skills', 'edu-abroad', 'edu-corporate', 'cb-ecommerce', 'cb-factory', 'cb-logistics', 'cb-payment', 'cb-brand']
|
|
|
|
api_slug_list = [s[0] for s in api_slugs]
|
|
in_api = [s for s in new_design_slugs if s in api_slug_list]
|
|
not_in_api = [s for s in new_design_slugs if s not in api_slug_list]
|
|
print(f'\n新设计子板块: 总{len(new_design_slugs)}, API中有{len(in_api)}, 缺失{len(not_in_api)}')
|
|
if not_in_api:
|
|
print(f'缺失: {not_in_api}')
|