feat(presegment): G2 PR-2 — presegment 워커 + 큐 배선 + range-clamp (deterministic ToC)

extract 前 presegment 스테이지: 전 문서 진입, 非PDF/단일은 무변 통과, '명확한 번들' PDF만 ToC(level-1) deterministic 분할. LLM 폴백은 PR-3. - presegment_worker: 보수적 게이트(pages>=60·자식>=5p·연속/단조/전범위·2<=N<=50) + 멱등 (lineage segmented_from 존재 시 수렴) + 자식=부모파일 공유(Option A)+range - queue_consumer: BATCH_SIZE/MAIN_QUEUE_STAGES/_load_workers + presegment->extract 전이, parent(번들원본)는 억제(자식이 직접 extract enqueue) - ingest(documents.py upload·file_watcher): 첫 stage extract->presegment - extract_worker/marker_worker: bundle_page_start/end 시 해당 범위만 추출/변환 (NULL=일반문서 byte-identical 무회귀 — 검수 확인) 코드 검수 완료(무회귀·full_path 스코프·NOT NULL 커버·py_compile). **미배포** — 실제 번들 PDF 처리 검증 후 배포(PR-3 LLM 폴백과 함께). Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-18 16:55:27 +09:00
parent d75fb7adaa
commit c3d5c33813
6 changed files with 486 additions and 25 deletions
@@ -67,21 +67,45 @@ def _postprocess_ocr(text: str) -> str:
    return text.strip()


-def _extract_pdf_pymupdf(file_path: Path) -> str:
-    """PyMuPDF fallback — 페이지 단위 스트리밍으로 대형 PDF도 저메모리 처리"""
+def _extract_pdf_pymupdf(
+    file_path: Path, start_page: int | None = None, end_page: int | None = None
+) -> str:
+    """PyMuPDF fallback — 페이지 단위 스트리밍으로 대형 PDF도 저메모리 처리.
+
+    G2 (PR-G2-2): start_page/end_page(1-based inclusive) 가 주어지면 그 범위만 추출
+    (번들 자식 doc = 부모 파일 공유 + 자기 page 범위). 둘 다 None = 전체(기존 동작 동일).
+    """
    import fitz
    text_parts = []
    with fitz.open(str(file_path)) as doc:
-        for page in doc:
-            text_parts.append(page.get_text())
+        if start_page is None and end_page is None:
+            for page in doc:
+                text_parts.append(page.get_text())
+        else:
+            # 1-based inclusive → 0-based range. 범위는 [0, page_count] 로 클램프(방어).
+            total = doc.page_count
+            lo = max(1, start_page or 1) - 1
+            hi = min(total, end_page or total)        # inclusive 끝 (0-based 마지막 인덱스 = hi-1)
+            for i in range(lo, hi):
+                text_parts.append(doc.load_page(i).get_text())
    return "\n".join(text_parts)


-def _get_pdf_page_count(file_path: Path) -> int:
-    """PDF 페이지 수 확인"""
+def _get_pdf_page_count(
+    file_path: Path, start_page: int | None = None, end_page: int | None = None
+) -> int:
+    """PDF 페이지 수 확인. G2: 범위가 주어지면 그 범위의 페이지 수(자식 doc 밀도 계산용).
+
+    둘 다 None = 전체 페이지 수(기존 동작 동일).
+    """
    import fitz
    with fitz.open(str(file_path)) as doc:
-        return len(doc)
+        total = len(doc)
+        if start_page is None and end_page is None:
+            return total
+        lo = max(1, start_page or 1)
+        hi = min(total, end_page or total)
+        return max(0, hi - lo + 1)


 async def _call_ocr(file_path: Path, is_image: bool, max_pages: int = 200) -> str | None:
@@ -310,6 +334,43 @@ async def process(document_id: int, session: AsyncSession) -> None:
        doc.extracted_at = datetime.now(timezone.utc)
        return

+    # ─── G2 (PR-G2-2): 번들 자식 PDF — 부모 파일 공유 + 자기 page 범위만 추출 ───
+    # kordoc 서비스는 page-range 파라미터가 없어 전체 파일을 파싱한다(자식엔 부적합) → kordoc
+    # 우회, PyMuPDF 로 [bundle_page_start, bundle_page_end] 범위만 추출. range OCR 은 본 PR 범위
+    # 밖(자식은 ToC 존재 = digital text layer 전제 → 대개 OCR 불필요). PyMuPDF 텍스트가 빈약해도
+    # 그대로 보존하고 사유를 남긴다.
+    if fmt == "pdf" and doc.bundle_page_start is not None and doc.bundle_page_end is not None:
+        if not full_path.exists():
+            raise FileNotFoundError(f"파일 없음: {full_path}")
+        start, end = doc.bundle_page_start, doc.bundle_page_end
+        try:
+            pymupdf_text = _extract_pdf_pymupdf(full_path, start, end)
+            page_count = _get_pdf_page_count(full_path, start, end)
+        except Exception as e:
+            logger.error(f"[pymupdf:child] {doc.file_path} pages={start}-{end} 실패: {e}")
+            raise
+
+        meta = doc.extract_meta or {}
+        meta["presegment_child_range"] = {"start_page": start, "end_page": end}
+        meta["pymupdf_chars"] = len(pymupdf_text.strip())
+        should, reason = _should_ocr(pymupdf_text, page_count)
+        if should:
+            # range OCR 미지원(후속 PR) — PyMuPDF 결과 유지 + 사유 기록(silent skip 아님).
+            meta["ocr_skip_reason"] = "presegment_child_range_ocr_unsupported"
+            meta["ocr_reason"] = reason
+            logger.warning(
+                f"[pymupdf:child] {doc.file_path} pages={start}-{end} "
+                f"OCR 필요({reason})하나 range OCR 미지원 → PyMuPDF 결과 유지"
+            )
+        doc.extracted_text = pymupdf_text.replace("\x00", "")
+        doc.extracted_at = datetime.now(timezone.utc)
+        doc.extractor_version = PYMUPDF_VERSION if pymupdf_text.strip() else None
+        doc.extract_meta = meta
+        logger.info(
+            f"[pymupdf:child] {doc.file_path} pages={start}-{end} ({len(pymupdf_text)}자)"
+        )
+        return
+
    # ─── kordoc 파싱 (HWP/HWPX/PDF) + PyMuPDF fallback + OCR ───
    if fmt in KORDOC_FORMATS:
        container_path = f"/documents/{doc.file_path}"