"""문제 사이트들의 HTML을 들여다보고 정확한 breadcrumb 셀렉터 찾기.""" import sys, io, json, re sys.stdout = io.TextIOWrapper(sys.stdout.buffer, encoding='utf-8') import requests, urllib3 urllib3.disable_warnings() HEADERS = { 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36', 'Accept': 'text/html,application/xhtml+xml,*/*;q=0.9', 'Accept-Language': 'ko-KR,ko;q=0.9', } # 사이트별 샘플 URL samples = { '인천광역시': 'https://www.incheon.go.kr/IC010101', # 인천소식 > 새소식 '교육부': 'https://www.moe.go.kr/sub/infoRenew.do?m=011602&page=011602&s=moe', '성평등가족부': 'https://www.mogef.go.kr/sp/geq/sp_geq_f001.do', # 정책정보 > 성평등 > X '외교부': 'https://www.mofa.go.kr/www/wpge/m_3435/contents.do', # 영사 > 여행 '과학기술정보통신부': 'https://www.msit.go.kr/contents/cont.do?sCode=user&mPid=13&mId=227', } s = requests.Session() s.headers.update(HEADERS) from bs4 import BeautifulSoup for site, url in samples.items(): print(f'\n========== {site} ==========') print(f'URL: {url}') try: r = s.get(url, verify=False, timeout=20) print(f'HTTP {r.status_code} / len {len(r.text)}') if r.status_code != 200: continue soup = BeautifulSoup(r.text, 'html.parser') # 가능한 breadcrumb 클래스 모두 candidates = [] for el in soup.find_all(['div', 'p', 'ul', 'ol', 'nav']): cls = ' '.join(el.get('class', [])) idv = el.get('id', '') kw = (cls + ' ' + idv).lower() if re.search(r'loc|path|bread|crumb|위치|where', kw): txt = el.get_text(' ', strip=True) if 5 < len(txt) < 250 and ('홈' in txt or '>' in txt or 'home' in txt.lower() or '·' in txt): candidates.append((cls or idv, txt[:200])) seen = set() for cls, txt in candidates[:10]: if cls in seen: continue seen.add(cls) print(f' [{cls}] -> {txt}') # title도 다시 title = soup.title.get_text(' ', strip=True) if soup.title else '' print(f' title: {title[:150]}') except Exception as e: print(f' ERR: {e}')