28 lines
907 B
Python
28 lines
907 B
Python
import re
|
|
import urllib.request
|
|
|
|
req = urllib.request.Request('https://www.zhuig.com/community', headers={'User-Agent': 'Mozilla/5.0', 'Cache-Control': 'no-cache'})
|
|
html = urllib.request.urlopen(req, timeout=15).read().decode('utf-8')
|
|
|
|
idx_q = html.find('全部板块')
|
|
idx_z = html.find('最新话题')
|
|
section = html[idx_q:idx_z] if idx_z > 0 else html[idx_q:idx_q+15000]
|
|
|
|
# 找 dangerous 内容
|
|
print(f'板块区域长度: {len(section)}')
|
|
print(f'\n板块区域前500字符:')
|
|
print(section[:500])
|
|
print(f'\n...')
|
|
print(f'\n板块区域后200字符:')
|
|
print(section[-200:])
|
|
|
|
# 找<a 出现
|
|
real_a = re.findall(r'<a\s', section)
|
|
print(f'\n真实<a>数: {len(real_a)}')
|
|
# 找href="/community/
|
|
href_a = re.findall(r'href="/community/[^"]+"', section)
|
|
print(f'href="/community/... 出现数: {len(href_a)}')
|
|
# 找RSC $ a
|
|
rsc_a = re.findall(r'\\"a\\"', section)
|
|
print(f'RSC \\"a\\" 组件数: {len(rsc_a)}')
|