import re import urllib.request # 重新拉取 req = urllib.request.Request('https://www.zhuig.com/community', headers={'User-Agent': 'Mozilla/5.0', 'Cache-Control': 'no-cache', 'Pragma': 'no-cache'}) html = urllib.request.urlopen(req, timeout=15).read().decode('utf-8') # 找"全部板块"区域 idx = html.find('全部板块') print(f'HTML size: {len(html)}, 全部板块 位置: {idx}') # 取 "全部板块" 之后的板块区域 section_start = html.find('全部板块') section_end = html.find('最新话题', section_start) section = html[section_start:section_end] if section_end > 0 else html[section_start:section_start+15000] print(f'板块区域长度: {len(section)} 字符') # 提取所有 链接 pattern = re.compile(r']*href="/community/([^"]+)"[^>]*>(.*?)', re.DOTALL) links = [] for m in pattern.finditer(section): slug = m.group(1) text = re.sub(r'<[^>]+>', ' ', m.group(2)) text = re.sub(r'\s+', ' ', text).strip() if text and len(text) < 30: links.append((slug, text)) print(f'\n=== 板块区域内的所有链接 (共{len(links)}个) ===') # 按slug排序去重 seen = set() for slug, text in links: if slug in seen: continue seen.add(slug) print(f' /community/{slug:25s} {text}') # 看父板块"电商零售"周围的HTML print('\n=== 电商零售(ecommerce) 周围的HTML片段 ===') m = re.search(r'.{200}电商零售.{500}', section) if m: print(m.group()[:800])