Files
zhuiguang-ai/scripts/check-render.py
T

42 lines
1.7 KiB
Python

import re
import urllib.request
req = urllib.request.Request('https://www.zhuig.com/community', headers={'User-Agent': 'Mozilla/5.0', 'Cache-Control': 'no-cache'})
html = urllib.request.urlopen(req, timeout=15).read().decode('utf-8')
# 找所有 <a> 链接的位置
all_a = list(re.finditer(r'<a[^>]*href="(/community/[^"]+)"', html))
print(f'整页 <a href="/community/..."> 总数: {len(all_a)}')
# 看这些链接在HTML中的位置分布
positions = [m.start() for m in all_a]
print(f'位置: {positions[:5]}...{positions[-5:] if len(positions) > 5 else ""}')
# 找 "全部板块" 位置
idx_quanbu = html.find('全部板块')
idx_zuixin = html.find('最新话题')
idx_luntan = html.find('论坛')
print(f'\n全部板块位置: {idx_quanbu}')
print(f'最新话题位置: {idx_zuixin}')
print(f'论坛位置: {idx_luntan}')
# 看 "全部板块" 之前有多少板块链接(即头部导航/侧边栏的)
before_count = sum(1 for p in positions if p < idx_quanbu)
in_section_count = sum(1 for p in positions if idx_quanbu < p < idx_zuixin)
after_count = sum(1 for p in positions if p > idx_zuixin)
print(f' "全部板块"之前: {before_count}个链接')
print(f' "全部板块"区域内: {in_section_count}个链接')
print(f' "最新话题"之后: {after_count}个链接')
# 找"全部板块"和"最新话题"之间的HTML
section = html[idx_quanbu:idx_zuixin] if idx_zuixin > 0 else ""
print(f'\n板块区域HTML大小: {len(section)}')
# 看里面有没有h3标题
h3_count = len(re.findall(r'<h3', section))
print(f'板块区域 <h3> 标签数: {h3_count}')
# 找h3的内容
for m in re.finditer(r'<h3[^>]*>(.*?)</h3>', section, re.DOTALL):
text = re.sub(r'<[^>]+>', '', m.group(1)).strip()
if text:
print(f' <h3>: {text}')