# -*- coding: utf-8 -*- """M검토 보류분(페이지→게시판 후보, URL/M불명) 재조사. 본문에 '실제 글목록'(상세링크≥3 또는 리스트DOM+페이징)이 있으면 진짜 게시판으로 확정. 검색창만 있고 목록이 없으면 페이지 유지(검색창 누출 거짓양성). 사용: python -X utf8 _재검수_mreview.py <기관...> [--write] """ import sys, os, re, json, shutil, importlib.util, time from concurrent.futures import ThreadPoolExecutor, as_completed import openpyxl HERE = os.path.dirname(os.path.abspath(__file__)) TEMP = os.path.join(HERE, '..', '_temp') MODULES = ['_chungnam_phase234_all.py', '_chungbuk_phase234_all.py', '_jeonbuk_phase234_all.py'] def load_sites(): sites, mods = {}, {} for f in MODULES: spec = importlib.util.spec_from_file_location(f[:-3], os.path.join(HERE, f)) m = importlib.util.module_from_spec(spec); spec.loader.exec_module(m) for k, v in m.SITES.items(): sites[k] = v; mods[k] = m return sites, mods LIST_SEL = ('table tbody tr, .board_list li, ul.bbs_list li, .bbs_list li, ' '.board_list tr, ul.list li, .list_wrap li, .gallery_list li, ' '.board tbody tr, .tbl_list tbody tr, .photo_list li') def make_fetch(M, cfg): if hasattr(M, 'make_session'): sess = M.make_session(weak_ssl=cfg.get('weak_ssl', False)) return lambda u: M.fetch(sess, u) return lambda u: M.fetch(u) def investigate(M, url, body_sel, fetchfn): soup = None for k in range(3): soup = fetchfn(url) if soup is not None: break time.sleep(0.5*(k+1)) if soup is None: return {'verdict': 'fail'} body = M.get_body(soup, body_sel) details = M.extract_detail_urls(body, url, limit=30) list_rows = [el for el in body.select(LIST_SEL) if el.find('a', href=True)] has_paging = bool(body.select('.pagination, .paging, nav.paging, .page_nav, .paginate')) txt = body.get_text(' ', strip=True) m = M.TOTAL_PAT.search(txt) or M.TOTAL_PAT_LOOSE.search(txt) total = int(m.group(1).replace(',', '')) if (m and m.group(1).replace(',','').isdigit()) else None is_board = (len(details) >= 3) or (len(list_rows) >= 3 and (has_paging or total is not None)) if is_board: mval = total if (total is not None and total <= 200000) else len(set(details)) or len(list_rows) return {'verdict': 'board', 'M': mval, 'detail': len(details), 'rows': len(list_rows), 'paging': has_paging, 'total': total} return {'verdict': 'page', 'detail': len(details), 'rows': len(list_rows), 'paging': has_paging} def run(name, cfg, M, write): jp = os.path.join(TEMP, f'_audit_{name}.json') d = json.load(open(jp, encoding='utf-8')) # 보류분: 페이지→게시판인데 apply가 적용 안 한 것 = 전부(여기서 재조사). apply가 이미 BOARD_URL로 적용한 건 제외. BOARD_URL = re.compile(r'selectBbsNttList|/bbs/|BBSMSTR|board/list|list\.buan|selectBoardList|selectCntrct|selectPric|/list\.do|mode=list|nttList', re.I) cand = [] for x in d['L_diffs']: if x['oldL']=='페이지' and x['newL']=='게시판': nm = x['newM'] or 0 try: nm=int(nm) except: nm=0 applied = (3 <= nm <= 200000) and bool(BOARD_URL.search(x.get('url',''))) if not applied: cand.append(x) body_sel = cfg['body_sel']; fetchfn = make_fetch(M, cfg) out = [] with ThreadPoolExecutor(max_workers=6) as ex: futs = {ex.submit(investigate, M, x['url'], body_sel, fetchfn): x for x in cand} for f in as_completed(futs): x = futs[f] try: r = f.result() except Exception: r = {'verdict':'fail'} out.append((x, r)) boards = [(x,r) for x,r in out if r['verdict']=='board'] pages = [(x,r) for x,r in out if r['verdict']=='page'] fails = [(x,r) for x,r in out if r['verdict']=='fail'] if write and boards: xlsx = cfg['xlsx'] bak = xlsx.replace('.xlsx','_backup_mreview전.xlsx') wb = openpyxl.load_workbook(xlsx); ws = wb.active if not os.path.exists(bak): shutil.copy(xlsx, bak) for x,r in boards: ws.cell(x['r'],12).value='게시판'; ws.cell(x['r'],13).value=r['M'] wb.save(xlsx) print(f"{name}: 후보{len(cand)} → 게시판확정{len(boards)} 페이지유지{len(pages)} 실패{len(fails)} {'[적용]' if write else '[DRY]'}") for x,r in sorted(boards, key=lambda t:t[0]['r'])[:30]: print(f" 게시판 r{x['r']} {(x['label'] or '')[:22]} M={r['M']} (상세{r['detail']}/목록{r['rows']}/페이징{r['paging']}) {x['url'][:60]}") return {'name':name,'board':len(boards),'page':len(pages),'fail':len(fails)} def main(): args = sys.argv[1:]; write = '--write' in args names = [a for a in args if not a.startswith('--')] sites, mods = load_sites() for n in names: if n not in sites: print(f'{n}: 없음'); continue run(n, sites[n], mods[n], write) if __name__ == '__main__': main()