25 lines
948 B
Python
25 lines
948 B
Python
import re
|
|
import urllib.request
|
|
|
|
req = urllib.request.Request('https://www.zhuig.com/community', headers={'User-Agent': 'Mozilla/5.0', 'Cache-Control': 'no-cache'})
|
|
html = urllib.request.urlopen(req, timeout=15).read().decode('utf-8')
|
|
|
|
# 找self.__next_f.push的script块
|
|
push_pattern = re.compile(r'self\.__next_f\.push\(\[1,"(.*?)"\]\)', re.DOTALL)
|
|
chunks = push_pattern.findall(html)
|
|
print(f'RSC push 块数: {len(chunks)}')
|
|
|
|
# 解码第一块
|
|
for i, chunk in enumerate(chunks):
|
|
# 反转义
|
|
decoded = chunk.encode().decode('unicode_escape')
|
|
# 找"电商零售"在不在
|
|
if '电商零售' in decoded or '平台电商' in decoded:
|
|
print(f'\n=== Chunk {i} (长{len(decoded)}) 包含板块数据 ===')
|
|
# 找电商零售位置
|
|
idx = decoded.find('电商零售')
|
|
print(f'电商零售位置: {idx}')
|
|
print(f'周围200字符: {decoded[max(0,idx-50):idx+400]}')
|
|
if i < 3:
|
|
break
|