import re import urllib.request req = urllib.request.Request('https://www.zhuig.com/community', headers={'User-Agent': 'Mozilla/5.0', 'Cache-Control': 'no-cache', 'Pragma': 'no-cache'}) html = urllib.request.urlopen(req, timeout=15).read().decode('utf-8') # 找板块区域 idx_q = html.find('全部板块') idx_z = html.find('最新话题') section = html[idx_q:idx_z] if idx_z > 0 else html[idx_q:idx_q+15000] # 统计 链接 a_count = len(re.findall(r']*href="/community/', section)) h3_count = len(re.findall(r']*>[^<]+', section)) print(f'板块区域 链接数: {a_count}') print(f'板块区域

标签数: {h3_count}') # 提取前10个 matches = list(re.finditer(r']*href="/community/([^"]+)"[^>]*>(.*?)', section, re.DOTALL))[:15] print('\n=== 板块区域前15个链接 ===') for m in matches: slug = m.group(1) text = re.sub(r'<[^>]+>', ' ', m.group(2)) text = re.sub(r'\s+', ' ', text).strip() print(f' /community/{slug:25s} | {text}')