"""코드 단위(unit) 추출. (계획서 §4.2) unit_type: FORM / METHOD / FUNCTION / MODULE / EVENT / CLASS_DEF / DECLARATION / MACRO unit_id : PROGRAM#INCLUDE#TYPE#NAME """ from __future__ import annotations import hashlib from dataclasses import dataclass, field from .statements import Statement EVENT_KEYWORDS = { "INITIALIZATION", "START-OF-SELECTION", "END-OF-SELECTION", "TOP-OF-PAGE", "END-OF-PAGE", "AT", # AT SELECTION-SCREEN..., AT LINE-SELECTION, AT USER-COMMAND "GET", # 논리DB GET "LOAD-OF-PROGRAM", } DECL_KEYWORDS = { "TYPES", "DATA", "CONSTANTS", "TABLES", "STATICS", "FIELD-SYMBOLS", "SELECT-OPTIONS", "PARAMETERS", "PARAMETER", "SELECTION-SCREEN", "RANGES", "CONTROLS", "CLASS-DATA", "INCLUDE", "TYPE-POOLS", "REPORT", "PROGRAM", "NODES", } SUB_CHUNK_LIMIT = 300 # 이 줄 수를 넘으면 최상위 IF/CASE/LOOP 경계로 서브청크 기록 @dataclass class Unit: unit_id: str program: str include: str unit_type: str name: str line_start: int line_end: int signature: str = "" header_comment: str = "" code_hash: str = "" loc: int = 0 stmt_indexes: list[int] = field(default_factory=list) # 이 unit 에 속한 문장 인덱스 sub_chunks: list[dict] = field(default_factory=list) def _event_name(up: list[str]) -> str | None: head = up[0] if head in {"INITIALIZATION", "START-OF-SELECTION", "END-OF-SELECTION", "TOP-OF-PAGE", "END-OF-PAGE", "LOAD-OF-PROGRAM"}: return head if head == "AT" and len(up) >= 2 and up[1] in {"SELECTION-SCREEN", "LINE-SELECTION", "USER-COMMAND"}: # AT NEW/END/FIRST/LAST 는 LOOP 제어문이므로 이벤트가 아니다 return " ".join(up[:4])[:60] if head == "GET" and len(up) == 2 and up[1] not in { "PARAMETER", "TIME", "CURSOR", "BADI", "REFERENCE", "PF-STATUS", "RUN" } and not up[1].startswith("'"): return f"GET {up[1]}" # 논리DB 이벤트 return None def extract_units(program: str, include: str, statements: list[Statement], source_lines: int) -> list[Unit]: units: list[Unit] = [] open_unit: Unit | None = None # FORM/FUNCTION/MODULE/MACRO/METHOD open_event: Unit | None = None open_class: Unit | None = None # CLASS_DEF (DEFINITION) in_class_impl: str | None = None # CLASS ... IMPLEMENTATION 의 클래스명 decl_stmts: list[int] = [] def uid(utype: str, name: str) -> str: return f"{program}#{include}#{utype}#{name.upper()}" def close_decl() -> None: nonlocal decl_stmts if decl_stmts: first, last = decl_stmts[0], decl_stmts[-1] u = Unit( unit_id=uid("DECLARATION", f"DECL_{statements[first].line_start}"), program=program, include=include, unit_type="DECLARATION", name=f"DECL_{statements[first].line_start}", line_start=statements[first].line_start, line_end=statements[last].line_end, stmt_indexes=list(decl_stmts), ) units.append(u) decl_stmts = [] def close_event(end_line: int) -> None: nonlocal open_event if open_event: open_event.line_end = end_line units.append(open_event) open_event = None for i, st in enumerate(statements): up = st.upper head = up[0] if up else "" # ---- 열려 있는 유닛 닫기 판단 ---- if open_unit: open_unit.stmt_indexes.append(i) end_kw = {"FORM": "ENDFORM", "FUNCTION": "ENDFUNCTION", "MODULE": "ENDMODULE", "METHOD": "ENDMETHOD", "MACRO": "END-OF-DEFINITION"}[open_unit.unit_type] if head == end_kw: open_unit.line_end = st.line_end units.append(open_unit) open_unit = None continue if open_class: open_class.stmt_indexes.append(i) if head == "ENDCLASS": open_class.line_end = st.line_end units.append(open_class) open_class = None continue # ---- 새 유닛 시작 ---- if head == "FORM" and len(up) >= 2: close_decl(); close_event(st.line_start - 1) open_unit = Unit(uid("FORM", up[1]), program, include, "FORM", up[1], st.line_start, st.line_end, signature=" ".join(st.tokens[2:])[:300], header_comment=st.comment_text, stmt_indexes=[i]) continue if head == "FUNCTION" and len(up) >= 2: close_decl(); close_event(st.line_start - 1) open_unit = Unit(uid("FUNCTION", up[1]), program, include, "FUNCTION", up[1], st.line_start, st.line_end, header_comment=st.comment_text, stmt_indexes=[i]) continue if head == "MODULE" and len(up) >= 2: close_decl(); close_event(st.line_start - 1) mode = up[2] if len(up) >= 3 and up[2] in {"OUTPUT", "INPUT"} else "" open_unit = Unit(uid("MODULE", f"{up[1]}_{mode}" if mode else up[1]), program, include, "MODULE", f"{up[1]} {mode}".strip(), st.line_start, st.line_end, header_comment=st.comment_text, stmt_indexes=[i]) continue if head == "DEFINE" and len(up) >= 2: close_decl(); close_event(st.line_start - 1) open_unit = Unit(uid("MACRO", up[1]), program, include, "MACRO", up[1], st.line_start, st.line_end, stmt_indexes=[i]) continue if head == "METHOD" and in_class_impl and len(up) >= 2: close_decl() name = f"{in_class_impl}~{up[1]}" open_unit = Unit(uid("METHOD", name), program, include, "METHOD", name, st.line_start, st.line_end, header_comment=st.comment_text, stmt_indexes=[i]) continue if head == "CLASS" and len(up) >= 3: # `CLASS lcl_x DEFINITION DEFERRED.` / `… DEFINITION LOAD.` 는 **한 줄 선언**이다 — # ENDCLASS 가 없다. 블록으로 열면 그 뒤 인클루드 전체가 CLASS_DEF 하나로 삼켜져 # TOP 의 선언이 통째로 사라진다 (실측: ZCO_ALV 의 GO_ALV1 이하 전부). if "DEFERRED" in up[2:] or "LOAD" in up[2:]: if open_event: open_event.stmt_indexes.append(i) open_event.line_end = st.line_end else: decl_stmts.append(i) continue if "DEFINITION" in up[2:4]: close_decl(); close_event(st.line_start - 1) open_class = Unit(uid("CLASS_DEF", up[1]), program, include, "CLASS_DEF", up[1], st.line_start, st.line_end, stmt_indexes=[i]) continue if "IMPLEMENTATION" in up[2:4]: close_decl(); close_event(st.line_start - 1) in_class_impl = up[1] continue if head == "ENDCLASS": in_class_impl = None continue ev = _event_name(up) if up else None if ev: close_decl(); close_event(st.line_start - 1) open_event = Unit(uid("EVENT", ev.replace(" ", "_")), program, include, "EVENT", ev, st.line_start, st.line_end, header_comment=st.comment_text, stmt_indexes=[i]) continue # ---- 유닛 밖 문장 ---- if open_event: open_event.stmt_indexes.append(i) open_event.line_end = st.line_end continue if head.split("-")[0].split(":")[0] in DECL_KEYWORDS or head in DECL_KEYWORDS: decl_stmts.append(i) continue # 그 밖의 최상위 문장(드묾)은 DECLARATION 블록에 편입하지 않고 무시하되 카운트는 파서 리포트에서 decl_stmts.append(i) # 파일 끝 정리 if open_unit: open_unit.line_end = statements[-1].line_end if statements else source_lines units.append(open_unit) if open_class: open_class.line_end = statements[-1].line_end if statements else source_lines units.append(open_class) close_event(statements[-1].line_end if statements else source_lines) close_decl() for u in units: u.loc = u.line_end - u.line_start + 1 return units def finalize_units(units: list[Unit], statements_by_include: dict[str, list[Statement]], code_by_include: dict[str, str]) -> None: """code_hash / sub_chunks 계산.""" for u in units: code = code_by_include.get(u.include, "") lines = code.split("\n") body = "\n".join(lines[u.line_start - 1 : u.line_end]) u.code_hash = hashlib.sha256(body.encode("utf-8")).hexdigest() if u.loc > SUB_CHUNK_LIMIT: u.sub_chunks = _sub_chunks(u, statements_by_include.get(u.include, [])) def _sub_chunks(u: Unit, statements: list[Statement]) -> list[dict]: """최상위 IF/CASE/LOOP/WHILE/DO 블록 경계로 unit 을 서브청크 목록으로 나눈다.""" opens = {"IF": "ENDIF", "CASE": "ENDCASE", "LOOP": "ENDLOOP", "WHILE": "ENDWHILE", "DO": "ENDDO"} boundaries: list[int] = [u.line_start] depth = 0 for idx in u.stmt_indexes: st = statements[idx] head = st.upper[0] if st.tokens else "" if head in opens: if depth == 0: boundaries.append(st.line_start) depth += 1 elif head in opens.values(): depth = max(0, depth - 1) if depth == 0: boundaries.append(st.line_end + 1) boundaries.append(u.line_end + 1) bounds = sorted(set(boundaries)) chunks = [] for a, b in zip(bounds, bounds[1:]): if b - a > 0: chunks.append({"line_start": a, "line_end": b - 1}) # 너무 잘게 쪼개지면 SUB_CHUNK_LIMIT 안쪽으로 병합 merged: list[dict] = [] for c in chunks: if merged and (c["line_end"] - merged[-1]["line_start"]) <= SUB_CHUNK_LIMIT: merged[-1]["line_end"] = c["line_end"] else: merged.append(dict(c)) return merged