全项目扫描修复: Docker数据卷修复+安全requireAdmin+12页SEO+API白名单+脚本超时+常量提取+假数据删除
This commit is contained in:
@@ -0,0 +1,41 @@
|
||||
import re
|
||||
import urllib.request
|
||||
|
||||
# 重新拉取
|
||||
req = urllib.request.Request('https://www.zhuig.com/community', headers={'User-Agent': 'Mozilla/5.0', 'Cache-Control': 'no-cache', 'Pragma': 'no-cache'})
|
||||
html = urllib.request.urlopen(req, timeout=15).read().decode('utf-8')
|
||||
|
||||
# 找"全部板块"区域
|
||||
idx = html.find('全部板块')
|
||||
print(f'HTML size: {len(html)}, 全部板块 位置: {idx}')
|
||||
|
||||
# 取 "全部板块" 之后的板块区域
|
||||
section_start = html.find('全部板块')
|
||||
section_end = html.find('最新话题', section_start)
|
||||
section = html[section_start:section_end] if section_end > 0 else html[section_start:section_start+15000]
|
||||
print(f'板块区域长度: {len(section)} 字符')
|
||||
|
||||
# 提取所有 <a> 链接
|
||||
pattern = re.compile(r'<a[^>]*href="/community/([^"]+)"[^>]*>(.*?)</a>', re.DOTALL)
|
||||
links = []
|
||||
for m in pattern.finditer(section):
|
||||
slug = m.group(1)
|
||||
text = re.sub(r'<[^>]+>', ' ', m.group(2))
|
||||
text = re.sub(r'\s+', ' ', text).strip()
|
||||
if text and len(text) < 30:
|
||||
links.append((slug, text))
|
||||
|
||||
print(f'\n=== 板块区域内的所有链接 (共{len(links)}个) ===')
|
||||
# 按slug排序去重
|
||||
seen = set()
|
||||
for slug, text in links:
|
||||
if slug in seen:
|
||||
continue
|
||||
seen.add(slug)
|
||||
print(f' /community/{slug:25s} {text}')
|
||||
|
||||
# 看父板块"电商零售"周围的HTML
|
||||
print('\n=== 电商零售(ecommerce) 周围的HTML片段 ===')
|
||||
m = re.search(r'.{200}电商零售.{500}', section)
|
||||
if m:
|
||||
print(m.group()[:800])
|
||||
Reference in New Issue
Block a user