import urllib.request req = urllib.request.Request('https://www.zhuig.com/community', headers={'User-Agent': 'Mozilla/5.0 (iPhone)', 'Cache-Control': 'no-cache', 'Pragma': 'no-cache'}) html = urllib.request.urlopen(req, timeout=15).read().decode('utf-8') # 保存完整HTML with open('/tmp/comm-fresh.html', 'w', encoding='utf-8') as f: f.write(html) # 提取构建ID import re build_id = re.search(r'BUILD_ID["\s:]+["\']([^"\']+)', html) print(f'HTML size: {len(html)}') print(f'BUILD_ID: {build_id.group(1) if build_id else "N/A"}') # 提取所有 _next/static 资源链接 static_links = sorted(set(re.findall(r'/_next/static/([^"\']+)', html))) print(f'\n静态资源: {len(static_links)}个') for s in static_links[:15]: print(f' /_next/static/{s}') # 检查"子板块是否在HTML里" new_subcats = ['ec-platform', 'ec-livestream', 'ec-supply', 'ec-dtc', 'ec-private', 'ec-group', 'ai-tools-app', 'ai-llm', 'ai-startup-forum', 'ai-saas', 'ai-hardware'] for s in new_subcats: if s in html: # 找包含这个slug的整段 idx = html.find(s) ctx = html[max(0, idx-50):idx+150] ctx_clean = re.sub(r'<[^>]+>', ' ', ctx) ctx_clean = re.sub(r'\s+', ' ', ctx_clean) print(f'\n [{s}] 出现: ...{ctx_clean[:200]}...') else: print(f'\n [{s}] NOT FOUND')