From 668ad6a3300c9d4c1e5bc9454d1a1a439293cf85 Mon Sep 17 00:00:00 2001 From: NaveenBuidl Date: Wed, 8 Apr 2026 22:34:55 +0200 Subject: [PATCH 1/5] Add minimal eval CI gate for contextual precision --- .github/workflows/eval-gate.yml | 73 ++++++++++++++++++++++++++ .gitignore | 3 +- eval/compare_runs.py | 92 +++++++++++++++++++++++++++++++++ 3 files changed, 167 insertions(+), 1 deletion(-) create mode 100644 .github/workflows/eval-gate.yml create mode 100644 eval/compare_runs.py diff --git a/.github/workflows/eval-gate.yml b/.github/workflows/eval-gate.yml new file mode 100644 index 0000000..2677da7 --- /dev/null +++ b/.github/workflows/eval-gate.yml @@ -0,0 +1,73 @@ +name: Eval Gate + +on: + pull_request: + workflow_dispatch: + inputs: + retrieval_k: + description: "Optional retrieval_k override (use 1 to run degraded config)" + required: false + default: "" + +jobs: + contextual-precision-gate: + runs-on: ubuntu-latest + timeout-minutes: 30 + steps: + - uses: actions/checkout@v4 + + - uses: actions/setup-python@v5 + with: + python-version: "3.11" + + - name: Install dependencies + run: pip install -r requirements.txt + + - name: Optionally override retrieval_k + if: ${{ github.event_name == 'workflow_dispatch' && github.event.inputs.retrieval_k != '' }} + run: | + python - <<'PY' + from pathlib import Path + import yaml + + config_path = Path("config.yaml") + data = yaml.safe_load(config_path.read_text(encoding="utf-8")) or {} + data["retrieval_k"] = int("${{ github.event.inputs.retrieval_k }}") + config_path.write_text(yaml.safe_dump(data, sort_keys=False), encoding="utf-8") + print(f"Using retrieval_k={data['retrieval_k']}") + PY + + - name: Run eval and enforce ContextualPrecisionMetric gate + env: + GROQ_API_KEY: ${{ secrets.GROQ_API_KEY }} + run: | + set -euo pipefail + + uvicorn app.main:app --host 127.0.0.1 --port 8000 > uvicorn.log 2>&1 & + UVICORN_PID=$! + trap 'kill $UVICORN_PID || true' EXIT + + for i in {1..30}; do + if curl -sf http://127.0.0.1:8000/docs > /dev/null; then + break + fi + sleep 2 + done + + python eval/run_eval.py | tee eval_output.txt + + python - <<'PY' + import re + from pathlib import Path + + threshold = 0.64 + text = Path("eval_output.txt").read_text(encoding="utf-8", errors="ignore") + match = re.search(r"ContextualPrecisionMetric: avg_score=([0-9.]+)", text) + if not match: + raise SystemExit("Could not find aggregate ContextualPrecisionMetric in eval output") + + score = float(match.group(1)) + print(f"ContextualPrecisionMetric avg_score={score:.3f} (threshold={threshold:.2f})") + if score < threshold: + raise SystemExit(f"CI gate failed: ContextualPrecisionMetric {score:.3f} < {threshold:.2f}") + PY \ No newline at end of file diff --git a/.gitignore b/.gitignore index 7db792b..b9493ed 100644 --- a/.gitignore +++ b/.gitignore @@ -6,4 +6,5 @@ chroma_db/ data/ *.log .chroma/ -.deepeval/ \ No newline at end of file +.deepeval/ +eval/run_eval_output_*.txt \ No newline at end of file diff --git a/eval/compare_runs.py b/eval/compare_runs.py new file mode 100644 index 0000000..924d434 --- /dev/null +++ b/eval/compare_runs.py @@ -0,0 +1,92 @@ +from __future__ import annotations + +import argparse +import re +from pathlib import Path + + +def read_text(path: Path) -> str: + raw = path.read_bytes() + for enc in ("utf-8", "utf-16", "utf-16-le", "utf-16-be"): + try: + return raw.decode(enc).replace("\x00", "") + except UnicodeDecodeError: + continue + return raw.decode("utf-8", errors="replace").replace("\x00", "") + + +def parse_sections(path: Path) -> tuple[dict[str, float], dict[str, float]]: + text = read_text(path) + lines = [line.strip() for line in text.splitlines()] + + aggregate: dict[str, float] = {} + categories: dict[str, float] = {} + + in_aggregate = False + in_categories = False + avg_re = re.compile(r"^-\s+(\w+):\s+avg_score=([^\s]+)") + + for line in lines: + if line == "Aggregate metric averages:": + in_aggregate = True + in_categories = False + continue + if line == "Simple average by category:": + in_categories = True + in_aggregate = False + continue + if not line: + continue + + m = avg_re.match(line) + if not m: + continue + + name, value = m.groups() + if value == "N/A": + continue + try: + score = float(value) + except ValueError: + continue + + if in_aggregate: + aggregate[name] = score + elif in_categories: + categories[name] = score + + return aggregate, categories + + +def print_table(title: str, baseline: dict[str, float], degraded: dict[str, float]) -> None: + keys = sorted(set(baseline) | set(degraded)) + print(f"\n{title}") + print("-" * len(title)) + print(f"{'name':35} {'baseline':>10} {'degraded':>10} {'delta':>10}") + for key in keys: + b = baseline.get(key) + d = degraded.get(key) + if b is None or d is None: + print(f"{key:35} {str(b):>10} {str(d):>10} {'N/A':>10}") + else: + print(f"{key:35} {b:10.3f} {d:10.3f} {d - b:+10.3f}") + + +def main() -> None: + parser = argparse.ArgumentParser(description="Compare baseline vs degraded eval outputs.") + parser.add_argument("baseline", type=Path, help="Path to baseline eval output file") + parser.add_argument("degraded", type=Path, help="Path to degraded eval output file") + args = parser.parse_args() + + base_agg, base_cat = parse_sections(args.baseline) + deg_agg, deg_cat = parse_sections(args.degraded) + + print(f"Baseline: {args.baseline}") + print(f"Degraded: {args.degraded}") + + print_table("Aggregate metric averages", base_agg, deg_agg) + print_table("Category averages", base_cat, deg_cat) + + +if __name__ == "__main__": + main() \ No newline at end of file From 00c894bffab7c3189773b249773f8afcebf0c976 Mon Sep 17 00:00:00 2001 From: NaveenBuidl Date: Thu, 9 Apr 2026 08:41:11 +0200 Subject: [PATCH 2/5] Add deepeval to requirements for CI --- requirements.txt | 1 + 1 file changed, 1 insertion(+) diff --git a/requirements.txt b/requirements.txt index 8594511..c0add6d 100644 --- a/requirements.txt +++ b/requirements.txt @@ -6,3 +6,4 @@ sentence-transformers groq python-dotenv pyyaml +deepeval From 815435a84cb7288b0925acd85cef1733738c6426 Mon Sep 17 00:00:00 2001 From: NaveenBuidl Date: Thu, 9 Apr 2026 09:02:08 +0200 Subject: [PATCH 3/5] Pass API keys to eval CI --- .github/workflows/eval-gate.yml | 1 + 1 file changed, 1 insertion(+) diff --git a/.github/workflows/eval-gate.yml b/.github/workflows/eval-gate.yml index 2677da7..5b3e830 100644 --- a/.github/workflows/eval-gate.yml +++ b/.github/workflows/eval-gate.yml @@ -39,6 +39,7 @@ jobs: - name: Run eval and enforce ContextualPrecisionMetric gate env: + OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} GROQ_API_KEY: ${{ secrets.GROQ_API_KEY }} run: | set -euo pipefail From 5ab4ade769d9d74d36b7a9b79c74a1b9cdab0a8d Mon Sep 17 00:00:00 2001 From: NaveenBuidl Date: Thu, 9 Apr 2026 09:13:22 +0200 Subject: [PATCH 4/5] Fail fast if app does not start in CI --- .github/workflows/eval-gate.yml | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/.github/workflows/eval-gate.yml b/.github/workflows/eval-gate.yml index 5b3e830..0043ce9 100644 --- a/.github/workflows/eval-gate.yml +++ b/.github/workflows/eval-gate.yml @@ -50,11 +50,19 @@ jobs: for i in {1..30}; do if curl -sf http://127.0.0.1:8000/docs > /dev/null; then + echo "App is up" + READY=1 break fi sleep 2 done + if [ "${READY:-0}" != "1" ]; then + echo "Uvicorn failed to start. Dumping log:" + cat uvicorn.log || true + exit 1 + fi + python eval/run_eval.py | tee eval_output.txt python - <<'PY' From 25d2d713f17d218bff30d6274d70ecf20144da46 Mon Sep 17 00:00:00 2001 From: NaveenBuidl Date: Thu, 9 Apr 2026 10:07:43 +0200 Subject: [PATCH 5/5] Make eval gate corpus path portable in CI with smoke corpus override --- .github/workflows/eval-gate.yml | 1 + app/config.py | 15 +++++++++++++-- corpus/ci_smoke/01_pricing_and_plans.md | 11 +++++++++++ corpus/ci_smoke/02_early_stage_program.md | 13 +++++++++++++ corpus/ci_smoke/03_scope_boundaries.md | 9 +++++++++ 5 files changed, 47 insertions(+), 2 deletions(-) create mode 100644 corpus/ci_smoke/01_pricing_and_plans.md create mode 100644 corpus/ci_smoke/02_early_stage_program.md create mode 100644 corpus/ci_smoke/03_scope_boundaries.md diff --git a/.github/workflows/eval-gate.yml b/.github/workflows/eval-gate.yml index 0043ce9..5841035 100644 --- a/.github/workflows/eval-gate.yml +++ b/.github/workflows/eval-gate.yml @@ -41,6 +41,7 @@ jobs: env: OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} GROQ_API_KEY: ${{ secrets.GROQ_API_KEY }} + EVALENS_CORPUS_PATH: corpus/ci_smoke run: | set -euo pipefail diff --git a/app/config.py b/app/config.py index 77eaef3..389a1f9 100644 --- a/app/config.py +++ b/app/config.py @@ -23,6 +23,13 @@ class Settings: groq_api_key: str +def _resolve_corpus_path(raw_path: str, config_dir: Path) -> str: + candidate = Path(raw_path).expanduser() + if not candidate.is_absolute(): + candidate = (config_dir / candidate).resolve() + return str(candidate) + + def load_settings(config_path: str = "config.yaml") -> Settings: cfg_file = Path(config_path) if not cfg_file.exists(): @@ -32,9 +39,13 @@ def load_settings(config_path: str = "config.yaml") -> Settings: cfg = yaml.safe_load(f) or {} groq_api_key = os.getenv("GROQ_API_KEY", "") + raw_corpus_path = os.getenv( + "EVALENS_CORPUS_PATH", + cfg.get("corpus_path", "D:/Evalens/corpus/intercom_external/raw_pdfs"), + ) return Settings( - corpus_path=cfg.get("corpus_path", "D:/Evalens/corpus/intercom_external/raw_pdfs"), + corpus_path=_resolve_corpus_path(raw_corpus_path, cfg_file.parent.resolve()), chunk_size=int(cfg.get("chunk_size", 1000)), chunk_overlap=int(cfg.get("chunk_overlap", 150)), retrieval_k=int(cfg.get("retrieval_k", 4)), @@ -43,4 +54,4 @@ def load_settings(config_path: str = "config.yaml") -> Settings: chroma_path=cfg.get("chroma_path", ".chroma"), collection_name=cfg.get("collection_name", "intercom_pdfs"), groq_api_key=groq_api_key, - ) + ) \ No newline at end of file diff --git a/corpus/ci_smoke/01_pricing_and_plans.md b/corpus/ci_smoke/01_pricing_and_plans.md new file mode 100644 index 0000000..e4f07c5 --- /dev/null +++ b/corpus/ci_smoke/01_pricing_and_plans.md @@ -0,0 +1,11 @@ +# Pricing and plans quick reference + +- There is **no Pro plan**. There is a **Pro add-on** for reporting. +- Pro add-on includes: CX Score, Topics Explorer, Trends, Recommendations, Monitors, and Custom Scorecards. +- Pro add-on pricing: $99/month base for up to 1,000 conversations; conversation-volume based (not seat-based). +- Fin AI Agent with Zendesk appears in two pricing representations in docs: + - $0.99 per outcome with commitments. + - $49/month includes 50 outcomes, then $0.99 per additional outcome. +- Copilot: all plans include limited usage for full-seat teammates (10 free conversations/month). +- Lite seats cannot use Copilot. +- Unlimited Copilot requires add-on pricing references of $29/agent/month (annual) or $35/seat/month (monthly). \ No newline at end of file diff --git a/corpus/ci_smoke/02_early_stage_program.md b/corpus/ci_smoke/02_early_stage_program.md new file mode 100644 index 0000000..5e8c3a1 --- /dev/null +++ b/corpus/ci_smoke/02_early_stage_program.md @@ -0,0 +1,13 @@ +# Early Stage program quick reference + +Eligibility requirements: +1. Startup has raised up to $10M in funding. +2. Startup has fewer than 15 employees. +3. Startup is not currently an Intercom customer. + +Clarifications: +- Existing paid customers (including annual plans) are not eligible. +- If someone only started a trial, they can still apply. +- Early Stage discount is only available on the Advanced plan. +- Early Stage discount does not apply to the Expert plan. +- If Expert features are needed, Expert is purchased at standard pricing. \ No newline at end of file diff --git a/corpus/ci_smoke/03_scope_boundaries.md b/corpus/ci_smoke/03_scope_boundaries.md new file mode 100644 index 0000000..21d6f5d --- /dev/null +++ b/corpus/ci_smoke/03_scope_boundaries.md @@ -0,0 +1,9 @@ +# Scope boundaries and abstention guidance + +This corpus covers product and plan information only. + +Out of scope examples: +- Intercom stock price is not included in this corpus. +- Platform uptime guarantee percentage is not included in this corpus. + +If asked these questions, answer that the indexed corpus does not contain that information. \ No newline at end of file