From c7270115975e0dd3359de1acd45f35d7234cd9d7 Mon Sep 17 00:00:00 2001 From: dnjsgkfka Date: Fri, 11 Sep 2026 17:06:05 +0900 Subject: [PATCH 1/5] =?UTF-8?q?ci:=20MLflow(DagsHub)=20=EB=B2=A4=EC=B9=98?= =?UTF-8?q?=EB=A7=88=ED=81=AC=20=EC=9E=90=EB=8F=99=ED=99=94=20=EC=9B=8C?= =?UTF-8?q?=ED=81=AC=ED=94=8C=EB=A1=9C=20=EC=B6=94=EA=B0=80?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .github/workflows/benchmark.yml | 73 +++++++++++++++++++++++++++++++++ .gitignore | 3 ++ benchmarks/measure_recall.py | 59 ++++++++++++++++++++++++-- pyproject.toml | 1 + 4 files changed, 133 insertions(+), 3 deletions(-) create mode 100644 .github/workflows/benchmark.yml diff --git a/.github/workflows/benchmark.yml b/.github/workflows/benchmark.yml new file mode 100644 index 0000000..69f799c --- /dev/null +++ b/.github/workflows/benchmark.yml @@ -0,0 +1,73 @@ +name: Benchmark + +# push할 때는 빠른 dev-only(20개 문서)로 자동 실행 + +on: + push: + branches: ["**"] + paths: + - "src/**" + - "benchmarks/measure_recall.py" + - "pyproject.toml" + workflow_dispatch: + inputs: + full: + description: "전체 90개 문서로 실행" + type: boolean + default: false + +permissions: + contents: read + +jobs: + benchmark: + runs-on: ubuntu-latest + timeout-minutes: 240 + + steps: + - uses: actions/checkout@v4 + + - uses: actions/setup-python@v5 + with: + python-version: "3.12" + cache: "pip" + cache-dependency-path: pyproject.toml + + - name: 패키지 설치 (evidence-chunker + dev/tracking extras + kaggle CLI) + run: | + pip install -e ".[dev,tracking]" + pip install kaggle + + - name: Kaggle에서 벤치마크 PDF/QA 데이터셋 설치 + env: + KAGGLE_USERNAME: ${{ secrets.KAGGLE_USERNAME }} + KAGGLE_KEY: ${{ secrets.KAGGLE_KEY }} + run: | + kaggle datasets download -d wonhalam/evidence-chunker-bench -p ./data --unzip + + - name: 데이터 경로 확인 + run: | + echo "pdfs: $(ls data/pdfs | wc -l)개" + echo "auto_qa: $(ls data/auto_qa | wc -l)개" + + - name: 벤치마크 실행 + MLflow(DagsHub)로 기록 + env: + MLFLOW_TRACKING_URI: ${{ secrets.MLFLOW_TRACKING_URI }} + MLFLOW_TRACKING_USERNAME: ${{ secrets.MLFLOW_TRACKING_USERNAME }} + MLFLOW_TRACKING_PASSWORD: ${{ secrets.MLFLOW_TRACKING_PASSWORD }} + PYTHONUTF8: "1" + TORCHDYNAMO_DISABLE: "1" + run: | + ARGS="--pdf-dir ./data/pdfs --qa-dir ./data/auto_qa --out-dir ./results --mlflow" + if [ "${{ github.event.inputs.full }}" != "true" ]; then + ARGS="$ARGS --dev-only" + fi + python benchmarks/measure_recall.py $ARGS + + - name: 결과 JSON을 Actions 아티팩트로 보관 + if: always() + uses: actions/upload-artifact@v4 + with: + name: benchmark-results-${{ github.run_number }} + path: results/ + retention-days: 30 diff --git a/.gitignore b/.gitignore index d2dddd3..95019dd 100644 --- a/.gitignore +++ b/.gitignore @@ -10,6 +10,9 @@ data/outputs/ # 보고서 (재생성 가능) reports/ +# MLflow +mlruns/ + # Python __pycache__/ *.pyc diff --git a/benchmarks/measure_recall.py b/benchmarks/measure_recall.py index c9a0474..762c9f5 100644 --- a/benchmarks/measure_recall.py +++ b/benchmarks/measure_recall.py @@ -28,6 +28,7 @@ import argparse import gc import json +import os import re from collections import Counter, defaultdict from pathlib import Path @@ -385,15 +386,35 @@ def _ci(n): def run(pdf_dir: Path, qa_dir: Path, out_dir: Path, dev_only: bool, max_pdfs: int | None, - parity_check: bool = False) -> None: + parity_check: bool = False, use_mlflow: bool = False) -> None: from sentence_transformers import SentenceTransformer import torch device = "cuda" if torch.cuda.is_available() else "cpu" + tag = "dev20" if dev_only else "full90" + + if use_mlflow: + import mlflow + if not os.environ.get("MLFLOW_TRACKING_URI"): + mlflow_db = (Path.cwd() / "mlflow.db").resolve() + mlflow.set_tracking_uri(f"sqlite:///{mlflow_db.as_posix()}") + mlflow.set_experiment("evidence-chunker-benchmark") + mlflow.start_run(run_name=tag) + mlflow.log_params({ + "tag": tag, + "bbox_threshold": BBOX_THRESHOLD, + "sim_threshold": SIM_THRESHOLD, + "embed_model": EMBED_MODEL_NAME, + "encode_batch": ENCODE_BATCH, + "min_doc_n": MIN_DOC_N, + "max_pdfs": max_pdfs, + }) pairs = pdf_qa_pairs(pdf_dir, qa_dir, dev_only, max_pdfs) if not pairs: print("[ERR] PDF-QA 쌍 없음") + if use_mlflow: + mlflow.end_run(status="FAILED") return import evidence_chunker @@ -413,6 +434,8 @@ def run(pdf_dir: Path, qa_dir: Path, out_dir: Path, dev_only: bool, max_pdfs: in if not rows: print("[ERR] 결과 없음") + if use_mlflow: + mlflow.end_run(status="FAILED") return N = len(rows) @@ -521,7 +544,6 @@ def blk(rs): "eu_em": round(_rate(rs, "e_em"), 4), "ci_halfwidth_pp": round(_ci(len(rs)), 2)} - tag = "dev20" if dev_only else "full90" summary = { "config": {"scope": tag, "qa_dir": str(qa_dir), "embed_model": EMBED_MODEL_NAME, "bbox_threshold": BBOX_THRESHOLD, "sim_threshold": SIM_THRESHOLD, @@ -564,6 +586,17 @@ def blk(rs): json.dumps(rows, indent=2, ensure_ascii=False), encoding="utf-8") print(f"\n저장: bench_{tag}.json / bench_{tag}_rows.json ({len(rows)} rows)") + if use_mlflow: + mlflow.log_metrics(summary["macro_average"]) + if summary.get("macro_average_min_n"): + mlflow.log_metrics({ + f"minN_{k}": v for k, v in summary["macro_average_min_n"].items() + if isinstance(v, (int, float)) + }) + mlflow.log_artifact(str(out_dir / f"bench_{tag}.json")) + mlflow.log_artifact(str(out_dir / f"bench_{tag}_rows.json")) + mlflow.end_run() + # =========================================================================== # 6. CLI @@ -579,12 +612,32 @@ def parse_args() -> argparse.Namespace: p.add_argument("--max-pdfs", type=int, default=None, help="추가 상한 (디버깅용)") p.add_argument("--parity-check", action="store_true", help="첫 문서에서 EvidenceChunker.build_corpus() 결과와 대조") + p.add_argument("--mlflow", action="store_true") + # 스윕 자동화 + p.add_argument("--bbox-threshold", type=float, default=None, help="BBOX_THRESHOLD") + p.add_argument("--sim-threshold", type=float, default=None, help="SIM_THRESHOLD") + p.add_argument("--embed-model", type=str, default=None, help="EMBED_MODEL_NAME") + p.add_argument("--encode-batch", type=int, default=None, help="ENCODE_BATCH") return p.parse_args() def main() -> None: + global BBOX_THRESHOLD, SIM_THRESHOLD, EMBED_MODEL_NAME, ENCODE_BATCH, CTX_WINDOW_PT + args = parse_args() - run(args.pdf_dir, args.qa_dir, args.out_dir, args.dev_only, args.max_pdfs, args.parity_check) + + if args.bbox_threshold is not None: + BBOX_THRESHOLD = args.bbox_threshold + CTX_WINDOW_PT = args.bbox_threshold # 두 값은 항상 같이 움직임 + if args.sim_threshold is not None: + SIM_THRESHOLD = args.sim_threshold + if args.embed_model is not None: + EMBED_MODEL_NAME = args.embed_model + if args.encode_batch is not None: + ENCODE_BATCH = args.encode_batch + + run(args.pdf_dir, args.qa_dir, args.out_dir, args.dev_only, args.max_pdfs, args.parity_check, + args.mlflow) if __name__ == "__main__": diff --git a/pyproject.toml b/pyproject.toml index ce1f4a0..7b4f4fa 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -24,6 +24,7 @@ similarity = ["sentence-transformers>=2.2"] langchain = ["langchain-core"] llamaindex = ["llama-index-core"] dev = ["pytest>=8", "sentence-transformers>=2.2", "llama-index-core"] +tracking = ["mlflow>=2.14"] [project.urls] Repository = "https://github.com/EvidenceChunker/Evidence-Chunker" From 46a539d2f8570a2d883f4ba0b01394c9a39b7d72 Mon Sep 17 00:00:00 2001 From: dnjsgkfka Date: Fri, 11 Sep 2026 17:51:02 +0900 Subject: [PATCH 2/5] =?UTF-8?q?ci:=20=EB=8D=B0=EC=9D=B4=ED=84=B0=20?= =?UTF-8?q?=EC=86=8C=EC=8A=A4=EB=A5=BC=20Kaggle=EC=97=90=EC=84=9C=20DagsHu?= =?UTF-8?q?b=20Storage=EB=A1=9C=20=EB=B3=80=EA=B2=BD?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .github/workflows/benchmark.yml | 11 +++-- scripts/download_bench_data.py | 76 +++++++++++++++++++++++++++++++++ 2 files changed, 81 insertions(+), 6 deletions(-) create mode 100644 scripts/download_bench_data.py diff --git a/.github/workflows/benchmark.yml b/.github/workflows/benchmark.yml index 69f799c..52fe4e5 100644 --- a/.github/workflows/benchmark.yml +++ b/.github/workflows/benchmark.yml @@ -33,17 +33,16 @@ jobs: cache: "pip" cache-dependency-path: pyproject.toml - - name: 패키지 설치 (evidence-chunker + dev/tracking extras + kaggle CLI) + - name: 패키지 설치 (evidence-chunker + dev/tracking extras) run: | pip install -e ".[dev,tracking]" - pip install kaggle - - name: Kaggle에서 벤치마크 PDF/QA 데이터셋 설치 + - name: DagsHub Storage에서 벤치마크 PDF/QA 받기 (private 버킷, 토큰 인증) env: - KAGGLE_USERNAME: ${{ secrets.KAGGLE_USERNAME }} - KAGGLE_KEY: ${{ secrets.KAGGLE_KEY }} + DAGSHUB_TOKEN: ${{ secrets.MLFLOW_TRACKING_PASSWORD }} run: | - kaggle datasets download -d wonhalam/evidence-chunker-bench -p ./data --unzip + pip install dagshub boto3 + python scripts/download_bench_data.py --out-dir ./data - name: 데이터 경로 확인 run: | diff --git a/scripts/download_bench_data.py b/scripts/download_bench_data.py new file mode 100644 index 0000000..6935a86 --- /dev/null +++ b/scripts/download_bench_data.py @@ -0,0 +1,76 @@ +""" +download_bench_data.py — DagsHub Storage 버킷(dnjsgkfka/Evidence-Chunker)에서 +pdfs/, auto_qa/ 를 받아온다. CI(GitHub Actions)에서 Kaggle 대신 쓰는 용도. + +환경변수 DAGSHUB_TOKEN 필요 (개인 액세스 토큰). + +사용법: + python scripts/download_bench_data.py --out-dir ./data +""" + +from __future__ import annotations + +import argparse +import os +import sys +from pathlib import Path + +import dagshub +from dagshub.upload.wrapper import get_repo_bucket_client + +REPO_FULL = "dnjsgkfka/Evidence-Chunker" +BUCKET_NAME = "Evidence-Chunker" # 버킷 이름 = 레포 이름 +PREFIXES = ["pdfs/", "auto_qa/"] + + +def parse_args() -> argparse.Namespace: + p = argparse.ArgumentParser(description=__doc__.strip().splitlines()[0]) + p.add_argument("--out-dir", type=Path, required=True, help="pdfs/, auto_qa/ 를 받을 루트 디렉토리") + return p.parse_args() + + +def main() -> None: + args = parse_args() + + token = os.environ.get("DAGSHUB_TOKEN") + if not token: + sys.exit("환경변수 DAGSHUB_TOKEN이 없습니다.") + + dagshub.auth.add_app_token(token) + s3 = get_repo_bucket_client(REPO_FULL) + + total = 0 + for prefix in PREFIXES: + local_dir = args.out_dir / prefix.rstrip("/") + local_dir.mkdir(parents=True, exist_ok=True) + + continuation_token = None + count = 0 + while True: + kwargs = {"Bucket": BUCKET_NAME, "Prefix": prefix, "MaxKeys": 1000} + if continuation_token: + kwargs["ContinuationToken"] = continuation_token + resp = s3.list_objects_v2(**kwargs) + + for obj in resp.get("Contents", []): + key = obj["Key"] + fname = key[len(prefix):] + if not fname: + continue + dest = local_dir / fname + s3.download_file(BUCKET_NAME, key, str(dest)) + count += 1 + + if resp.get("IsTruncated"): + continuation_token = resp.get("NextContinuationToken") + else: + break + + print(f"[{prefix}] {count}개 파일 다운로드 완료 -> {local_dir}") + total += count + + print(f"\n총 {total}개 파일 다운로드 완료") + + +if __name__ == "__main__": + main() From a3745377778d6b823738672a1176efb6d64fdb9a Mon Sep 17 00:00:00 2001 From: dnjsgkfka Date: Sat, 12 Sep 2026 02:09:56 +0900 Subject: [PATCH 3/5] =?UTF-8?q?ci:=20=EC=9B=8C=ED=81=AC=ED=94=8C=EB=A1=9C?= =?UTF-8?q?=20=ED=8A=B8=EB=A6=AC=EA=B1=B0=20path=EC=97=90=20scripts/**,=20?= =?UTF-8?q?workflow=20=EC=B6=94=EA=B0=80?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .github/workflows/benchmark.yml | 2 ++ 1 file changed, 2 insertions(+) diff --git a/.github/workflows/benchmark.yml b/.github/workflows/benchmark.yml index 52fe4e5..e96bba7 100644 --- a/.github/workflows/benchmark.yml +++ b/.github/workflows/benchmark.yml @@ -8,7 +8,9 @@ on: paths: - "src/**" - "benchmarks/measure_recall.py" + - "sripts/**" - "pyproject.toml" + - ".github/workflows/benchmark.yml" workflow_dispatch: inputs: full: From 0780153b8599505ae5c8653f32c2126ac9494ba3 Mon Sep 17 00:00:00 2001 From: dnjsgkfka Date: Wed, 16 Sep 2026 10:13:21 +0900 Subject: [PATCH 4/5] =?UTF-8?q?ci:=20=EB=B2=A4=EC=B9=98=EB=A7=88=ED=81=AC?= =?UTF-8?q?=20=EC=9B=8C=ED=81=AC=ED=94=8C=EB=A1=9C=203=EB=8B=A8=EA=B3=84?= =?UTF-8?q?=EB=A1=9C=20=EB=B6=84=EB=A6=AC=20+=20QA=20=EC=9C=A0=ED=98=95?= =?UTF-8?q?=EB=B3=84=20MLflow=20=EB=A1=9C=EA=B9=85=20=EC=B6=94=EA=B0=80?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .github/workflows/benchmark.yml | 36 ++++++++++++---- .gitignore | 2 +- scripts/summarize_results.py | 73 +++++++++++++++++++++++++++++++++ 3 files changed, 102 insertions(+), 9 deletions(-) create mode 100644 scripts/summarize_results.py diff --git a/.github/workflows/benchmark.yml b/.github/workflows/benchmark.yml index e96bba7..924663d 100644 --- a/.github/workflows/benchmark.yml +++ b/.github/workflows/benchmark.yml @@ -1,14 +1,16 @@ name: Benchmark -# push할 때는 빠른 dev-only(20개 문서)로 자동 실행 +# 3단계로 나눠서 돈다 (CPU 러너라 무거운 걸 매번 돌리면 너무 오래 걸림): +# - main이 아닌 브랜치에 push -> smoke (5개 문서, 몇 분) : 코드가 안 깨졌는지만 빠르게 확인 +# - main에 push -> dev-only (20개 문서, ~1시간) : 회귀 체크 +# - workflow_dispatch (full: true) -> 전체 90개 문서 (수동 트리거 전용, 여러 시간 소요) on: push: - branches: ["**"] paths: - "src/**" - "benchmarks/measure_recall.py" - - "sripts/**" + - "scripts/**" - "pyproject.toml" - ".github/workflows/benchmark.yml" workflow_dispatch: @@ -51,6 +53,20 @@ jobs: echo "pdfs: $(ls data/pdfs | wc -l)개" echo "auto_qa: $(ls data/auto_qa | wc -l)개" + - name: 실행 범위 결정 (smoke / dev / full) + id: scope + run: | + if [ "${{ github.event_name }}" = "workflow_dispatch" ] && [ "${{ github.event.inputs.full }}" = "true" ]; then + echo "args=--mlflow" >> "$GITHUB_OUTPUT" + echo "label=full (90개)" >> "$GITHUB_OUTPUT" + elif [ "${{ github.ref }}" = "refs/heads/main" ]; then + echo "args=--mlflow --dev-only" >> "$GITHUB_OUTPUT" + echo "label=dev-only (20개)" >> "$GITHUB_OUTPUT" + else + echo "args=--mlflow --dev-only --max-pdfs 5" >> "$GITHUB_OUTPUT" + echo "label=smoke (5개)" >> "$GITHUB_OUTPUT" + fi + - name: 벤치마크 실행 + MLflow(DagsHub)로 기록 env: MLFLOW_TRACKING_URI: ${{ secrets.MLFLOW_TRACKING_URI }} @@ -59,11 +75,15 @@ jobs: PYTHONUTF8: "1" TORCHDYNAMO_DISABLE: "1" run: | - ARGS="--pdf-dir ./data/pdfs --qa-dir ./data/auto_qa --out-dir ./results --mlflow" - if [ "${{ github.event.inputs.full }}" != "true" ]; then - ARGS="$ARGS --dev-only" - fi - python benchmarks/measure_recall.py $ARGS + echo "실행 범위: ${{ steps.scope.outputs.label }}" + python benchmarks/measure_recall.py \ + --pdf-dir ./data/pdfs --qa-dir ./data/auto_qa --out-dir ./results \ + ${{ steps.scope.outputs.args }} + + - name: 결과 요약을 Actions 탭에 바로 표시 + if: always() + run: | + python scripts/summarize_results.py --results-dir ./results --label "${{ steps.scope.outputs.label }}" - name: 결과 JSON을 Actions 아티팩트로 보관 if: always() diff --git a/.gitignore b/.gitignore index 95019dd..fb761c2 100644 --- a/.gitignore +++ b/.gitignore @@ -1,4 +1,4 @@ -# 모델 파일 (용량 큼 — 로컬 다운로드) +# 모델 파일 models/ # 테스트 PDF diff --git a/scripts/summarize_results.py b/scripts/summarize_results.py new file mode 100644 index 0000000..a13bdc5 --- /dev/null +++ b/scripts/summarize_results.py @@ -0,0 +1,73 @@ +""" +summarize_results.py — results/bench_*.json 을 읽어서 GitHub Actions 의 +Step Summary(Actions 탭에서 바로 보이는 마크다운)로 출력한다. + +사용법 (CI 안에서): + python scripts/summarize_results.py --label "dev-only (20개)" +""" + +from __future__ import annotations + +import argparse +import glob +import json +import os + + +def parse_args() -> argparse.Namespace: + p = argparse.ArgumentParser(description=__doc__.strip().splitlines()[0]) + p.add_argument("--results-dir", default="results", help="bench_*.json 이 있는 디렉토리") + p.add_argument("--label", default="", help="이번 실행 범위 라벨 (smoke/dev/full 등)") + return p.parse_args() + + +def main() -> None: + args = parse_args() + + files = sorted(glob.glob(os.path.join(args.results_dir, "bench_*.json"))) + files = [f for f in files if not f.endswith("_rows.json")] + summary_path = os.environ.get("GITHUB_STEP_SUMMARY") + + lines = ["# 벤치마크 결과\n"] + if args.label: + lines.append(f"실행 범위: **{args.label}**\n") + + if not files: + lines.append("결과 파일을 찾지 못했습니다 (실행이 중간에 실패했을 수 있어요).\n") + + for f in files: + data = json.load(open(f, encoding="utf-8")) + overall = data.get("overall") or {} + lines.append(f"\n## {os.path.basename(f)}\n") + lines.append("| 지표 | baseline | EU | 문항 수 |") + lines.append("|---|---|---|---|") + lines.append( + f"| Recall | {overall.get('baseline_recall', '-')} | " + f"{overall.get('eu_recall', '-')} | {overall.get('n', '-')} |" + ) + lines.append( + f"| EM | {overall.get('baseline_em', '-')} | " + f"{overall.get('eu_em', '-')} | {overall.get('n', '-')} |" + ) + + by_type = data.get("by_type") or {} + type_blocks = {t: b for t, b in by_type.items() if b} + if type_blocks: + lines.append("\n**유형별 EM (baseline → EU)**\n") + lines.append("| 유형 | baseline EM | EU EM | n |") + lines.append("|---|---|---|---|") + for t, blk in type_blocks.items(): + lines.append( + f"| {t} | {blk.get('baseline_em', '-')} | " + f"{blk.get('eu_em', '-')} | {blk.get('n', '-')} |" + ) + + text = "\n".join(lines) + "\n" + print(text) + if summary_path: + with open(summary_path, "a", encoding="utf-8") as fh: + fh.write(text) + + +if __name__ == "__main__": + main() From 635c8bc3625165912cf634aa0f3e26201c7582b9 Mon Sep 17 00:00:00 2001 From: dnjsgkfka Date: Wed, 16 Sep 2026 10:13:38 +0900 Subject: [PATCH 5/5] =?UTF-8?q?fix:=20CI=EC=8B=9D=20=EC=88=98=EC=A0=95?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- benchmarks/measure_recall.py | 91 +++++++++++++++++++++++++++++++++++- 1 file changed, 90 insertions(+), 1 deletion(-) diff --git a/benchmarks/measure_recall.py b/benchmarks/measure_recall.py index 762c9f5..1f7df9d 100644 --- a/benchmarks/measure_recall.py +++ b/benchmarks/measure_recall.py @@ -18,6 +18,9 @@ 주의: normalize_for_em/has_token/em_hit은 benchmarks/generate_qa_docling.py의 동일 함수와 반드시 같은 규칙을 유지해야 한다 — 어긋나면 answer_spec이 의미를 잃는다. +통계: _ci()는 참고용 단순 근사치. 유의성 판단은 paired_statistics()의 문서 단위 +클러스터 부트스트랩 CI(주 지표) + McNemar 검정(보조 지표, 질문 간 독립 가정)을 쓴다. + 사용법: python measure_recall.py --pdf-dir ./data/pdfs --qa-dir ./auto_qa --out-dir ./results python measure_recall.py --pdf-dir ./data/pdfs --qa-dir ./auto_qa --out-dir ./results --dev-only @@ -28,11 +31,14 @@ import argparse import gc import json +import math import os import re from collections import Counter, defaultdict from pathlib import Path +import numpy as np + # --------------------------------------------------------------------------- # 하이퍼파라미터 — 라이브러리 기본값과 동일하게 유지 (sweep은 별도 실험) # --------------------------------------------------------------------------- @@ -381,10 +387,72 @@ def _line(label, rows, width=28): def _ci(n): - """95% CI 반폭(pp) — 표본이 작을 때 관측된 갭이 실제로 유의한지 판단하는 기준.""" + """95% CI 반폭(pp) — 표본이 작을 때 관측된 갭이 실제로 유의한지 판단하는 기준. + + 단순 이항분포 최대분산(p=0.5) 근사(Wald)라, 같은 질문에 대한 + baseline/EU 쌍대비교 구조나 문서 내 상관은 반영하지 않는다. 엄밀한 + 유의성 판단에는 아래 paired_statistics()를 쓸 것 — 이 함수는 빠른 + 참고용 오차범위로만 남겨둔다. + """ return 1.96 * 0.5 / (n ** 0.5) * 100 if n else float("inf") +def paired_statistics(doc_ids, baseline, treatment, repeats: int = 50000, seed: int = 20260915) -> dict: + """문서 단위 클러스터 부트스트랩 CI + 보조 McNemar 검정. + + _ci()와 달리 같은 문서에서 나온 질문들을 묶어서(문서를 리샘플링 단위로 + 삼아) 차이의 95% percentile CI를 구한다 — 문서 내 질문 간 상관을 + 반영하는 방식. McNemar는 "baseline만 맞음 vs EU만 맞음"의 비대칭을 + 검정하는 보조 지표로 덧붙이되, 문서 간 의존성은 보정하지 않는다는 + 가정을 명시한다. + + baseline/treatment: 질문별 0/1(또는 bool) 정오답 배열. doc_ids와 길이가 + 같아야 하며, 같은 인덱스가 같은 질문을 가리켜야 한다(쌍대비교 전제). + """ + b = np.asarray(list(baseline), dtype=np.int64) + e = np.asarray(list(treatment), dtype=np.int64) + doc_ids = list(doc_ids) + if len(doc_ids) != len(b) or len(b) != len(e) or not len(b): + raise ValueError("Paired vectors must be nonempty and equal length.") + + grouped: dict = defaultdict(lambda: [0, 0]) # doc_id -> [n_questions, sum(e-b)] + for d, delta in zip(doc_ids, e - b): + grouped[d][0] += 1 + grouped[d][1] += int(delta) + a = np.asarray([grouped[d] for d in sorted(grouped)], dtype=np.int64) + + ci = None + if len(a) >= 2: + rng, values = np.random.default_rng(seed), [] + for start in range(0, repeats, 2048): + idx = rng.integers(0, len(a), size=(min(2048, repeats - start), len(a))) + total = a[idx].sum(axis=1) + values.append(100 * total[:, 1] / total[:, 0]) + ci = np.quantile(np.concatenate(values), [.025, .975]).tolist() + + b_only = int(((b == 1) & (e == 0)).sum()) + e_only = int(((b == 0) & (e == 1)).sum()) + discordant = b_only + e_only + chi2 = max(abs(b_only - e_only) - 1, 0) ** 2 / discordant if discordant else 0.0 + p = math.erfc(math.sqrt(chi2 / 2)) if discordant else 1.0 + + return { + "n_questions": len(b), "n_documents": len(a), + "baseline": float(b.mean()), "treatment": float(e.mean()), + "difference_pp": float(100 * (e - b).mean()), + "cluster_bootstrap_ci95_pp": ci, + "bootstrap": {"unit": "document", "statistic": "micro_rate_difference", + "method": "percentile", "repeats": repeats, "seed": seed}, + "mcnemar_supplementary": { + "method": "chi_square_continuity_corrected", "chi2": chi2, "p_value": p, + "baseline_only": b_only, "treatment_only": e_only, + "assumption": "질문 간 독립 가정 — 문서 내 의존성은 보정하지 않음(보조 지표)", + }, + "ci_note": ("문서를 독립 표집 단위로 취급한 근사치이며, LLM 생성 답변 정확도의 CI가 아님" + if ci is not None else "문서 2개 미만이라 CI 계산 불가"), + } + + def run(pdf_dir: Path, qa_dir: Path, out_dir: Path, dev_only: bool, max_pdfs: int | None, parity_check: bool = False, use_mlflow: bool = False) -> None: from sentence_transformers import SentenceTransformer @@ -504,6 +572,17 @@ def _mline(label, m, n): print(f"\n [real_driver] both_right={rd['both_right']} " f"baseline_win_eu_lose={rd['baseline_win_eu_lose']} " f"eu_win_baseline_lose={rd['eu_win_baseline_lose']} both_wrong={rd['both_wrong']}") + + paired_em = paired_statistics([r["doc_id"] for r in rows], + [r["b_em"] for r in rows], [r["e_em"] for r in rows]) + ci = paired_em["cluster_bootstrap_ci95_pp"] + ci_str = f"[{ci[0]:+.1f}, {ci[1]:+.1f}]pp" if ci else "N/A(문서<2)" + mc = paired_em["mcnemar_supplementary"] + print(f"\n [EM 통계 검정] 문서 단위 부트스트랩 차이 {paired_em['difference_pp']:+.1f}pp " + f"95% CI {ci_str}") + print(f" [McNemar 보조] baseline만 정답 {mc['baseline_only']} EU만 정답 {mc['treatment_only']} " + f"chi2={mc['chi2']:.2f} p={mc['p_value']:.2e}") + fails = Counter(r["fail_reason"] for r in rows if r["fail_reason"]) if fails: print(f"\n EU 실패 사유 상위") @@ -573,6 +652,7 @@ def blk(rs): "eu_em": round(mbig["e_em"], 4)} if mbig else None), "doc_question_counts": {d: len(v) for d, v in by_doc.items()}, "real_driver": dict(rd), + "paired_statistics_em": paired_em, "fail_reasons": dict(fails), "pipeline": {"n_eu": sum(r["n_eu"] for r in results), "n_split": ts, "dedup_removed": tc, "hybrid_before_dedup": tb, @@ -593,6 +673,15 @@ def blk(rs): f"minN_{k}": v for k, v in summary["macro_average_min_n"].items() if isinstance(v, (int, float)) }) + # QA 유형별(cell_value / table_about / context_dependent) 지표도 + # DagsHub에서 바로 비교할 수 있게 별도 metric으로 남긴다. + for t, blk_v in summary.get("by_type", {}).items(): + if not blk_v: + continue + mlflow.log_metrics({ + f"type_{t}_{k}": v for k, v in blk_v.items() + if isinstance(v, (int, float)) + }) mlflow.log_artifact(str(out_dir / f"bench_{tag}.json")) mlflow.log_artifact(str(out_dir / f"bench_{tag}_rows.json")) mlflow.end_run()