Files
zhuiguang-ai/scripts/debug-rsc.py
T

34 lines
1.2 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import re
import urllib.request
req = urllib.request.Request('https://www.zhuig.com/community', headers={'User-Agent': 'Mozilla/5.0', 'Cache-Control': 'no-cache'})
html = urllib.request.urlopen(req, timeout=15).read().decode('utf-8')
# 板块区域附近
idx_q = html.find('全部板块')
idx_z = html.find('最新话题')
section = html[idx_q:idx_z] if idx_z > 0 else html[idx_q:idx_q+15000]
# 找"<a " 出现位置(不是RSC payload里的"a"组件,是真实HTML)
real_a = re.findall(r'<a\s', section)
real_a_in_quotes = re.findall(r'\\"a\\",', section)
print(f'真实<a>标签: {len(real_a)}')
print(f'RSC \\"a\\" 组件: {len(real_a_in_quotes)}')
# 看section里"$" RSC payload 数量
dollar_count = section.count('"$"')
print(f'RSC payload "$" 数量: {dollar_count}')
# 找第一个 RSC "a" 组件
m = re.search(r'\["\$","a",[^]]+\]', section)
if m:
print(f'\n第一个RSC a组件示例:\n{m.group()[:500]}')
# 看板块区域最后部分(应该是RSC结束+可能HTML)
print(f'\n板块区域最后200字符:')
print(section[-300:])
# 板块区域有多少个 [$,"a" 出现
rsc_a = re.findall(r'\["\$","a"', section)
print(f'\nRSC "a" 组件数: {len(rsc_a)}')