Files
zhuiguang-ai/scripts/final-check.py
T

37 lines
1.2 KiB
Python

import re
import urllib.request
req = urllib.request.Request('https://www.zhuig.com/community', headers={'User-Agent': 'Mozilla/5.0', 'Cache-Control': 'no-cache'})
html = urllib.request.urlopen(req, timeout=15).read().decode('utf-8')
print(f'HTML total: {len(html)} bytes')
a_tags = len(re.findall(r'<a\s', html))
print(f'真实 <a> 标签: {a_tags}')
h2 = len(re.findall(r'<h2', html))
h3 = len(re.findall(r'<h3', html))
print(f'<h2>: {h2} <h3>: {h3}')
# 找字面平台电商
m = re.search(r'<h3[^>]*>[^<]*平台电商[^<]*</h3>', html)
print(f'字面<平台电商> HTML: {bool(m)}')
# 板块区域
idx_q = html.find('全部板块')
idx_z = html.find('最新话题')
section = html[idx_q:idx_z] if idx_z > 0 else html[idx_q:idx_q+15000]
print(f'板块区域长度: {len(section)}')
a_pat = r'<a\s'
h3_pat = r'<h3'
a_count = len(re.findall(a_pat, section))
h3_count = len(re.findall(h3_pat, section))
print(f'板块区域 <a 数量: {a_count}')
print(f'板块区域 <h3> 数量: {h3_count}')
# 找板块标题h2
m2 = re.search(r'<h2[^>]*>全部板块</h2>', section)
print(f'板块标题 h2 全部板块 存在: {bool(m2)}')
# 看section的前500字符
print(f'\n板块区域前800字符:')
print(section[:800])