Files
zhuiguang-ai/scripts/debug-section.py
T

30 lines
1.2 KiB
Python

import re
import urllib.request
req = urllib.request.Request('https://www.zhuig.com/community', headers={'User-Agent': 'Mozilla/5.0', 'Cache-Control': 'no-cache'})
html = urllib.request.urlopen(req, timeout=15).read().decode('utf-8')
idx_q = html.find('全部板块')
idx_z = html.find('最新话题')
section = html[idx_q:idx_z] if idx_z > 0 else html[idx_q:idx_q+15000]
# 找"电商零售"在section中的位置
kw_pos = section.find('电商零售')
print(f'电商零售在section位置: {kw_pos}')
if kw_pos >= 0:
print(f'周围300字符: {section[max(0,kw_pos-100):kw_pos+200]}')
# 找全部的板块数据标识 - slug 出现的位置
slugs = ['ec-platform', 'ec-livestream', 'ec-supply', 'ec-dtc', 'ec-private', 'ec-group', 'ai-tools-app', 'ai-llm', 'ai-startup-forum', 'ai-saas', 'ai-hardware']
print(f'\n=== 各slug在section中是否出现 ===')
for s in slugs:
p = section.find(f'"{s}"')
p2 = section.find(f'/{s}')
print(f' {s:25s} "slug"-模式:{p} /slug-模式:{p2}')
# 找父板块的name - "电商零售" 看看上下文
print(f'\n=== 在section里查"电商零售"上下文 ===')
if '电商零售' in section:
idx = section.find('电商零售')
print(section[max(0,idx-50):idx+300])