34 lines
1.3 KiB
Python
34 lines
1.3 KiB
Python
import urllib.request
|
|
|
|
req = urllib.request.Request('https://www.zhuig.com/community', headers={'User-Agent': 'Mozilla/5.0 (iPhone)', 'Cache-Control': 'no-cache', 'Pragma': 'no-cache'})
|
|
html = urllib.request.urlopen(req, timeout=15).read().decode('utf-8')
|
|
|
|
# 保存完整HTML
|
|
with open('/tmp/comm-fresh.html', 'w', encoding='utf-8') as f:
|
|
f.write(html)
|
|
|
|
# 提取构建ID
|
|
import re
|
|
build_id = re.search(r'BUILD_ID["\s:]+["\']([^"\']+)', html)
|
|
print(f'HTML size: {len(html)}')
|
|
print(f'BUILD_ID: {build_id.group(1) if build_id else "N/A"}')
|
|
|
|
# 提取所有 _next/static 资源链接
|
|
static_links = sorted(set(re.findall(r'/_next/static/([^"\']+)', html)))
|
|
print(f'\n静态资源: {len(static_links)}个')
|
|
for s in static_links[:15]:
|
|
print(f' /_next/static/{s}')
|
|
|
|
# 检查"子板块是否在HTML里"
|
|
new_subcats = ['ec-platform', 'ec-livestream', 'ec-supply', 'ec-dtc', 'ec-private', 'ec-group', 'ai-tools-app', 'ai-llm', 'ai-startup-forum', 'ai-saas', 'ai-hardware']
|
|
for s in new_subcats:
|
|
if s in html:
|
|
# 找包含这个slug的整段
|
|
idx = html.find(s)
|
|
ctx = html[max(0, idx-50):idx+150]
|
|
ctx_clean = re.sub(r'<[^>]+>', ' ', ctx)
|
|
ctx_clean = re.sub(r'\s+', ' ', ctx_clean)
|
|
print(f'\n [{s}] 出现: ...{ctx_clean[:200]}...')
|
|
else:
|
|
print(f'\n [{s}] NOT FOUND')
|