# -*- coding: utf-8 -*- """335~443행 분해 계획 산출(읽기전용). - 미분해 F섹션(type1 탭 URL이 시트에 1개=자기자신만 존재) → 그 type1 탭들로 분해 - 이미 분해된 잎 행의 깊은 중첩(type3가 별도 URL, 시트에 없음) → 그 type3로 분해 - type3가 #nav 앵커면 분해 안 함(M=앵커수) 재귀로 깊은 단계까지. 결과: 분해 잡 목록 + 총 증가행수. """ import re, json, warnings from urllib.parse import urlparse import requests from bs4 import BeautifulSoup import openpyxl warnings.filterwarnings('ignore') XLSX = r'D:\01.프로젝트\DB수집\작업파일\광역_사이트맵\충청남도\2.공주시\충청남도_공주시.xlsx' BASE = 'https://www.gongju.go.kr' H = {'User-Agent': 'Mozilla/5.0 Chrome/120 Safari/537.36'} LO, HI = 335, 443 _cache = {} def full(h): return h if h.startswith('http') else BASE + h def norm(u): """시트 멤버십 비교용 정규화: host 제거, path(+bbs id)만""" p = urlparse(u if u.startswith('http') else BASE + u) path = p.path return path def get(u): u = full(u) if u in _cache: return _cache[u] try: x = requests.get(u, headers=H, timeout=25, verify=False) x.encoding = x.apparent_encoding or 'utf-8' html = x.text except Exception: html = '' _cache[u] = html return html def tab_groups(html): s = BeautifulSoup(html, 'html.parser') t1 = [] t3 = [] for ul in s.find_all('ul'): cls = ' '.join(ul.get('class') or []) if 'tab-ul' not in cls.lower(): continue items = [(a.get_text(strip=True), a.get('href', '')) for a in ul.find_all('a') if a.get_text(strip=True) and a.get('href')] if not items: continue if 'type1' in cls: t1 = items elif not t3: t3 = items return t1, t3 def board_count(html): s = BeautifulSoup(html, 'html.parser') el = s.select_one('.program--count strong') if el: m = re.sub(r'[^0-9]', '', el.get_text()) return int(m) if m else None return None def is_board(u): return bool(re.search(r'list\.do|/bbs/|BBSMSTR', u)) def leaf_info(name, u): """잎 1개의 (L,M) 결정 + 그 잎의 #nav 앵커수 반영""" fu = full(u) html = get(fu) if is_board(u): c = board_count(html) return {'name': name, 'url': fu, 'L': '게시판', 'M': c if c is not None else 0} # 페이지: 자기 type3가 #nav면 M=앵커수 _, t3 = tab_groups(html) navs = [h for t, h in t3 if h.startswith('#')] if len(navs) >= 2: return {'name': name, 'url': fu, 'L': '페이지', 'M': len(navs)} return {'name': name, 'url': fu, 'L': '페이지', 'M': 1} def url_type3(u): """페이지의 type3 별도URL 탭들(없으면 [])""" if is_board(u): return [] _, t3 = tab_groups(get(full(u))) return [(t, full(h)) for t, h in t3 if not h.startswith('#')] def expand_nested(name, u, sheet_urls, depth=0): """깊이 2 평탄 분해(공주 탭 최대 2단계). 자식이 url-type3를 가지면 1단계만 더 펼침.""" grand = url_type3(u) if depth == 0 else [] if len(grand) >= 2: return {'name': name, 'url': full(u), 'children': [leaf_info(t, h) for t, h in grand]} return leaf_info(name, u) def main(): wb = openpyxl.load_workbook(XLSX) ws = wb.active sheet_urls = set() for r in range(3, 445): u = ws.cell(r, 11).value if isinstance(u, str) and u.startswith('http'): sheet_urls.add(norm(u)) jobs = [] for r in range(LO, HI + 1): u = ws.cell(r, 11).value if not (isinstance(u, str) and u.startswith('http')): continue html = get(u) t1, t3 = tab_groups(html) t1_urls = [(t, full(h)) for t, h in t1 if not h.startswith('#')] present = sum(1 for t, h in t1_urls if norm(h) in sheet_urls) url_t3 = [(t, full(h)) for t, h in t3 if not h.startswith('#')] label = (ws.cell(r, 6).value or '') if ws.cell(r, 7).value: label += ' > ' + str(ws.cell(r, 7).value) if ws.cell(r, 8).value: label += ' > ' + str(ws.cell(r, 8).value) # 미분해 F섹션: type1 URL>=2 이고 시트에 자기 1개만 → type1 자식(각자 type3 1단계 더) if len(t1_urls) >= 2 and present <= 1: children = [expand_nested(t, h, sheet_urls, depth=0) for t, h in t1_urls] jobs.append(('SECTION', r, label, children)) # 이미 분해된 G-잎의 깊은 type3 중첩 → type3 자식(잎) elif len(url_t3) >= 2: children = [leaf_info(t, h) for t, h in url_t3] jobs.append(('NEST', r, label, children)) # 요약 def count_leaves(nodes): n = 0 for nd in nodes: if nd.get('children'): n += count_leaves(nd['children']) else: n += 1 return n total_add = 0 print('=== 분해 계획 (335~443) ===') for kind, r, label, children in jobs: leaves = count_leaves(children) add = leaves - 1 # 기존 1행 대체 total_add += add print('\n[%s] r%d %s → 잎 %d (기존1, +%d)' % (kind, r, label, leaves, add)) def pr(nodes, ind=1): for nd in nodes: tag = ('게시판%s' % nd['M']) if nd.get('L') == '게시판' else ('페이지M%s' % nd.get('M', '')) if nd.get('children'): print(' ' * ind + '▸ %s' % nd['name']) pr(nd['children'], ind + 1) else: print(' ' * ind + '- %s [%s]' % (nd['name'], tag)) pr(children) print('\n총 분해 잡 %d개, 총 증가 행수 +%d (최종 데이터행 ~ %d)' % (len(jobs), total_add, 443 + total_add)) json.dump([{'kind': k, 'row': r, 'label': l, 'children': c} for k, r, l, c in jobs], open(r'D:\01.프로젝트\DB수집\_스크립트\_split_plan.json', 'w', encoding='utf-8'), ensure_ascii=False, indent=1) if __name__ == '__main__': main()