import re with open('/tmp/comm.html', encoding='utf-8') as f: html = f.read() slugs = sorted(set(re.findall(r'/community/([a-zA-Z0-9_-]+)', html))) print('板块slug数:', len(slugs)) for s in slugs: print(' ', s)