import re import urllib.request req = urllib.request.Request('https://www.zhuig.com/community', headers={'User-Agent': 'Mozilla/5.0', 'Cache-Control': 'no-cache'}) html = urllib.request.urlopen(req, timeout=15).read().decode('utf-8') idx_q = html.find('全部板块') idx_z = html.find('最新话题') section = html[idx_q:idx_z] if idx_z > 0 else html[idx_q:idx_q+15000] # 找"电商零售"在section中的位置 kw_pos = section.find('电商零售') print(f'电商零售在section位置: {kw_pos}') if kw_pos >= 0: print(f'周围300字符: {section[max(0,kw_pos-100):kw_pos+200]}') # 找全部的板块数据标识 - slug 出现的位置 slugs = ['ec-platform', 'ec-livestream', 'ec-supply', 'ec-dtc', 'ec-private', 'ec-group', 'ai-tools-app', 'ai-llm', 'ai-startup-forum', 'ai-saas', 'ai-hardware'] print(f'\n=== 各slug在section中是否出现 ===') for s in slugs: p = section.find(f'"{s}"') p2 = section.find(f'/{s}') print(f' {s:25s} "slug"-模式:{p} /slug-模式:{p2}') # 找父板块的name - "电商零售" 看看上下文 print(f'\n=== 在section里查"电商零售"上下文 ===') if '电商零售' in section: idx = section.find('电商零售') print(section[max(0,idx-50):idx+300])