# -*- coding: utf-8 -*- """서산 본문탭(div.tab_menu>ul.tab_button) 탐지 스캔(읽기전용). 각 L=페이지 행 URL을 열어 탭그룹 탐지. 조건: 탭>=2, 모든 href 실제링크, 현재행 URL이 탭집합에 포함. 탭그룹을 시그니처로 dedup해 보고. type: contents.do=페이지, selectBbsNttList.do=게시판. """ import openpyxl, urllib.request, ssl, re, html as ht, sys from urllib.parse import urlparse, parse_qs ctx=ssl.create_default_context(); ctx.check_hostname=False; ctx.verify_mode=ssl.CERT_NONE F='충청남도_서산시.xlsx' BASE='https://www.seosan.go.kr' def get(url): req=urllib.request.Request(url, headers={'User-Agent':'Mozilla/5.0'}) return urllib.request.urlopen(req,context=ctx,timeout=20).read().decode('utf-8','replace') def norm(u): """path+정렬된 query 로 정규화(도메인 제거).""" p=urlparse(u) q=parse_qs(p.query) qs='&'.join(f'{k}={q[k][0]}' for k in sorted(q)) return f'{p.path}?{qs}' def find_tabs(html): m=re.search(r'
(.*?)
', html, re.S) if not m: return [] block=m.group(1) if 'tab_button' not in block: return [] tabs=re.findall(r']*)>\s*]*>([^<]+)', block) out=[] for cls,href,txt in tabs: href=ht.unescape(href).strip() if href.startswith('#') or href.lower().startswith('javascript') or not href: continue out.append({'on':'on' in cls,'href':href,'label':txt.strip(), 'type':'게시판' if 'selectBbsNttList' in href else '페이지'}) return out def main(): wb=openpyxl.load_workbook(F); ws=wb.active last=max(r for r in range(3,ws.max_row+1) if ws.cell(r,2).value not in (None,'')) # excel URL 집합 excel_urls={} for r in range(3,last+1): u=ws.cell(r,11).value if u and 'seosan.go.kr' in str(u): excel_urls.setdefault(norm(str(u)),[]).append(r) groups={} scanned=0 for r in range(3,last+1): L=ws.cell(r,12).value; u=ws.cell(r,11).value if not (u and 'seosan.go.kr/www' in str(u)): continue if str(L) not in ('페이지','게시판'): continue try: d=get(str(u)) except: continue scanned+=1 tabs=find_tabs(d) if len(tabs)<2: continue nset={norm(BASE+t['href']) for t in tabs} if norm(str(u)) not in nset: continue # 자기 탭그룹 아님 sig='|'.join(sorted(nset)) if sig in groups: continue # 엑셀에 없는 탭 개수 missing=[t for t in tabs if norm(BASE+t['href']) not in excel_urls] groups[sig]={'anchor_row':r,'tabs':tabs,'missing':len(missing)} print(f'스캔 {scanned}행 / 탭그룹 {len(groups)}개') for sig,g in groups.items(): ar=g['anchor_row'] lab=ws.cell(ar,6).value or ws.cell(ar,5).value print(f"\n[앵커 행{ar} '{lab}'] 탭{len(g['tabs'])} 신규{g['missing']}") for t in g['tabs']: inx=excel_urls.get(norm(BASE+t['href'])) print(f" {'[ON]' if t['on'] else ' '} {t['type']:3} {t['label'][:24]:24} {t['href'][:46]} {'엑셀행'+str(inx) if inx else '★신규'}") if __name__=='__main__': main()