From 202d31ea06630886ec021ccb9ce9ded38caf9f49 Mon Sep 17 00:00:00 2001 From: Alicia Frame Date: Thu, 2 Jul 2026 16:10:57 -0400 Subject: [PATCH 1/2] Add analyze_auto_evals.py: retrieve and analyze auto-generated per-step evals Foundry attaches an OpenAI Evals object to every fine-tuning job and runs it every eval_interval steps (default 5). The per-step grader pass-rate curve is an early signal of whether a job is learning, well before the final model exists. - scripts/analyze_auto_evals.py: prints the per-step pass-rate curve with trend diagnosis, and dumps per-sample grader results for any step to JSONL. Uses common.get_clients; small output_items page size to avoid the API timeout. - references/auto-evals.md: object model, API/SDK usage, curve interpretation, and the large-page timeout gotcha. - SKILL.md / README.md: register the script and reference; add the timeout to Common Errors. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- Skills/README.md | 2 + Skills/SKILL.md | 5 + Skills/references/auto-evals.md | 80 ++++++++++ Skills/scripts/analyze_auto_evals.py | 209 +++++++++++++++++++++++++++ 4 files changed, 296 insertions(+) create mode 100644 Skills/references/auto-evals.md create mode 100644 Skills/scripts/analyze_auto_evals.py diff --git a/Skills/README.md b/Skills/README.md index ef4234b..c657ee5 100644 --- a/Skills/README.md +++ b/Skills/README.md @@ -82,6 +82,7 @@ Skills/ │ ├── deployment-formats.md # Model format, SKU, and version mapping │ ├── evaluation-methodology.md # Eval rubric design and grader types │ ├── training-curve-analysis.md # Reading training logs and curves (SFT + RFT) +│ ├── auto-evals.md # Reading auto-generated per-step evals (early progress signal) │ ├── grader-design.md # RFT grader design (type selection, partial credit, calibration) │ ├── foundry-cli.md # azd ai finetuning CLI reference │ ├── vision-fine-tuning.md # Image/video fine-tuning (gpt-4o, gpt-4.1) @@ -104,6 +105,7 @@ Skills/ │ ├── monitor_training.py # Poll a running job until completion │ ├── calibrate_grader.py # RFT grader threshold calibration │ ├── check_training.py # Training curve analysis and checkpoints +│ ├── analyze_auto_evals.py # Retrieve/analyze auto-generated per-step evals │ ├── deploy_model.py # Deploy via ARM REST API │ ├── evaluate_model.py # LLM judge evaluation │ ├── convert_dataset.py # Format conversion (SFT↔DPO↔RFT) diff --git a/Skills/SKILL.md b/Skills/SKILL.md index 40c5279..5e74381 100644 --- a/Skills/SKILL.md +++ b/Skills/SKILL.md @@ -50,6 +50,7 @@ Read the relevant reference file before performing any step: | `references/deployment-formats.md` | Deploying a fine-tuned model | | `references/evaluation-methodology.md` | Designing an eval rubric | | `references/training-curve-analysis.md` | Reading training logs and curves | +| `references/auto-evals.md` | Reading the auto-generated per-step evals for early progress signal | | `references/foundry-cli.md` | Using the `azd ai finetuning` CLI for submit/deploy | | `references/vision-fine-tuning.md` | Fine-tuning with image data (gpt-4o, gpt-4.1) | | `references/cost-management.md` | Training costs, hosting tiers, budget planning | @@ -74,6 +75,7 @@ Reusable Python scripts in `scripts/`. Each is self-contained with inline docume | `calibrate_grader.py` | Run base model through your RFT grader to find optimal pass_threshold | | `generate_distillation_data.py` | Generate training data from a teacher model for distillation (legacy custom-script approach) | | `check_training.py` | Pull training curves, detect overfitting, list checkpoints | +| `analyze_auto_evals.py` | Retrieve and analyze the **auto-generated evals** Foundry attaches to a job — prints the per-step pass-rate curve (an early signal of whether the job is learning, refreshed every `eval_interval` steps) and can dump per-sample grader results for any step to JSONL | | `deploy_model.py` | Deploy fine-tuned models via ARM REST API | | `cleanup.py` | List and delete old deployments, files, and pending jobs to reclaim quota | | `evaluate_model.py` | Run held-out eval with 2-dimension LLM judge | @@ -114,6 +116,8 @@ Reusable Python scripts in `scripts/`. Each is self-contained with inline docume | Submit SFT job | `python scripts/submit_training.py --model gpt-4.1-mini --training-file train.jsonl --validation-file val.jsonl --type sft` | | Monitor job | `python scripts/monitor_training.py --job-id ftjob-xxx` | | Analyze curves | `python scripts/check_training.py --job-id ftjob-xxx` | +| Auto-eval pass-rate curve (early signal) | `python scripts/analyze_auto_evals.py --job-id ftjob-xxx` | +| Dump per-sample auto-eval results for a step | `python scripts/analyze_auto_evals.py --job-id ftjob-xxx --run-id evalrun-xxx --dump results.jsonl` | | Deploy model | `python scripts/deploy_model.py --model-id ft:gpt-4.1-mini:... --name my-eval` | | Evaluate model | `python scripts/evaluate_model.py --deployment-name my-eval --test-file test.jsonl` | | Auto fine-tune | `python scripts/auto_finetune.py auto --data data.jsonl --description "task" --model gpt-4.1-mini` | @@ -130,6 +134,7 @@ Reusable Python scripts in `scripts/`. Each is self-contained with inline docume | Content safety block at deployment | PII-dense training data | Remove problematic document types | | "BadRequestForDependentService" | Deployment still warming up | Wait 5+ minutes after deployment creation | | Queue stuck ("jobs ahead") | Standard tier capacity exhausted | Cancel and resubmit on `developerTier` or `globalStandard` | +| Eval `output_items` request hangs / times out | `limit=100` on `/evals/{id}/runs/{run}/output_items` returns 120s+ (each item carries the full sample) | Use a small page size (`analyze_auto_evals.py` uses `limit=20`) | # Rules diff --git a/Skills/references/auto-evals.md b/Skills/references/auto-evals.md new file mode 100644 index 0000000..82c348d --- /dev/null +++ b/Skills/references/auto-evals.md @@ -0,0 +1,80 @@ +# Auto-Generated Evals (Early Progress Signal) + +Azure AI Foundry automatically attaches an **OpenAI Evals** object to every fine-tuning job +and runs it every `eval_interval` steps (default: **every 5 steps**). Each run scores the +validation set with the job's grader, so the pass-rate curve across steps is an **early +signal** of whether the job is learning — visible long before the final `fine_tuned_model` +exists, and available for SFT, DPO, and RFT jobs. + +This is complementary to `training-curve-analysis.md`: + +| Source | Metric | Best for | +|--------|--------|----------| +| Result CSV (`check_training.py`) | `valid_loss`, token accuracy | Overfitting, checkpoint selection | +| Auto-evals (`analyze_auto_evals.py`) | grader **pass rate** per step | Early "is it learning?" signal, per-sample failure inspection | + +## Object Model + +``` +fine_tuning.job.eval -> eval_ (one eval per job) + └─ runs (one per eval step) -> evalrun_ ("Step 5", "Step 10", ...) + └─ output_items (one per validation sample) (per-sample grader score + I/O) +``` + +## Retrieving via the API + +Everything lives under the project `/openai/v1` endpoint with Bearer auth. + +| Goal | Endpoint | +|------|----------| +| Find the eval id | `GET /fine_tuning/jobs/{job_id}` → `.eval` | +| Rollup across all steps | `GET /evals/{eval_id}` | +| Per-step pass/fail curve | `GET /evals/{eval_id}/runs?limit=100` → `.data[].result_counts` | +| One step's summary | `GET /evals/{eval_id}/runs/{run_id}` → `.per_testing_criteria_results` | +| Per-sample scores + model output | `GET /evals/{eval_id}/runs/{run_id}/output_items` | + +Python SDK (`openai>=1.40`): + +```python +job = client.fine_tuning.jobs.retrieve(job_id) +eval_id = job.eval +runs = client.evals.runs.list(eval_id, limit=100) # one per step +items = client.evals.runs.output_items.list(run_id, eval_id=eval_id, limit=20) +``` + +Note the argument order: `runs.list(eval_id, ...)` but `output_items.list(run_id, *, eval_id=...)`. + +## Using the Script + +```bash +# Per-step pass-rate curve + trend diagnosis (the early signal) +python scripts/analyze_auto_evals.py --job-id ftjob-xxx + +# Dump every per-sample result for one step to JSONL (inputs, outputs, tool calls, scores) +python scripts/analyze_auto_evals.py --job-id ftjob-xxx --run-id evalrun-yyy --dump results.jsonl +``` + +`--base-url` / `--api-key` fall back to `OPENAI_BASE_URL` / `AZURE_OPENAI_API_KEY`, or to +Foundry SDK / `DefaultAzureCredential` when no key is given (see `scripts/common.py`). + +## Reading the Curve + +- **Rising pass rate** → the job is learning. Even a noisy climb (e.g. 1% → 12% → 23%) is a + good early signal well before the job finishes. +- **Flat near zero** → the grader may be miscalibrated (nothing can pass), the task may be + too hard, or hyperparameters may be off. Inspect a run's `output_items` to see what the + model actually produced, then revisit `grader-design.md`. +- **Rising then falling** → possible over-training or reward hacking. Inspect a late run's + per-sample outputs and compare against `reward-hacking-prevention.md`. +- **Rising then plateauing** → returns are diminishing; more steps may not help. + +Because these evals refresh every few steps, you can catch a doomed run (flat/near-zero +pass rate) early and cancel it instead of paying for the full job. + +## Gotcha: Large `output_items` Pages Time Out + +Each `output_item` carries the full sample — prompt, model output, tool calls, and grader +result — so the payload is large. Requesting `limit=100` on the `output_items` endpoint +routinely hangs past the 120s server timeout and returns nothing. Use a small page size +(the script uses `limit=20`); the SDK auto-paginates as you iterate the cursor. Expect a +full dump of a few hundred samples to take several minutes. diff --git a/Skills/scripts/analyze_auto_evals.py b/Skills/scripts/analyze_auto_evals.py new file mode 100644 index 0000000..b8d9022 --- /dev/null +++ b/Skills/scripts/analyze_auto_evals.py @@ -0,0 +1,209 @@ +# /// script +# dependencies = [ +# "openai>=1.40", +# "azure-identity", +# "azure-ai-projects", +# ] +# /// +""" +analyze_auto_evals.py — Retrieve and analyze the auto-generated evals attached to a fine-tuning job. + +Azure AI Foundry automatically attaches an OpenAI **Evals** object to every fine-tuning +job and runs it every `eval_interval` steps (default: every 5 steps). Each run scores the +validation set with the job's grader, so the pass-rate curve across steps is an **early +signal** of whether the job is learning — long before the final `fine_tuned_model` exists. + +Object chain: + fine_tuning.job.eval -> eval_ + └─ runs (one per eval step) -> evalrun_ + └─ output_items (one per validation sample) -> per-sample grader score + +Usage: + # Per-step pass-rate curve + trend analysis (the early signal): + python analyze_auto_evals.py --job-id ftjob-abc123 + + # Dump every per-sample result for one run to JSONL (inputs, outputs, scores): + python analyze_auto_evals.py --job-id ftjob-abc123 --run-id evalrun_xyz --dump items.jsonl + +Connection (see common.py). Preferred: project /v1/ endpoint + key. + --base-url https://.openai.azure.com/openai/v1 (or OPENAI_BASE_URL) + --api-key (or AZURE_OPENAI_API_KEY) +Falls back to Foundry SDK / DefaultAzureCredential when no key is given. +""" +import json +import os +import sys + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +from common import HelpOnErrorParser, get_clients + + +def _step_num(name): + """Parse the trailing step number out of a run name like 'Step 115'.""" + try: + return int(str(name).strip().split()[-1]) + except (ValueError, IndexError): + return -1 + + +def _get_eval_id(job): + """The eval id lives on job.eval; fall back to model_extra for SDK version drift.""" + eval_id = getattr(job, "eval", None) + if not eval_id and getattr(job, "model_extra", None): + eval_id = job.model_extra.get("eval") + return eval_id + + +def curve(client, job_id): + """Print the per-step pass-rate curve and a short trend diagnosis.""" + job = client.fine_tuning.jobs.retrieve(job_id) + print(f"Job: {job.id}") + print(f" Status: {job.status}") + print(f" Fine-tuned model: {job.fine_tuned_model}") + + eval_id = _get_eval_id(job) + if not eval_id: + print("\n No eval attached to this job yet. Auto-evals appear once the first " + "eval step completes (default: after step 5).") + return + print(f" Eval: {eval_id}") + + # One run per eval step. Iterating the cursor page auto-paginates on long jobs. + runs = list(client.evals.runs.list(eval_id, limit=100)) + + if not runs: + print("\n No eval runs yet — check back after the first eval interval.") + return + + rows = [] + for r in runs: + rc = r.result_counts + total = rc.total or 0 + rate = (rc.passed / total * 100) if total else 0.0 + rows.append((_step_num(r.name), r.name, r.id, rc.passed, rc.failed, total, rate, r.status)) + rows.sort(key=lambda x: x[0]) + + print(f"\n Auto-eval pass-rate curve ({len(rows)} runs):") + print(f" {'Step':>6} {'Passed':>8} {'Total':>7} {'Pass %':>8} Run ID") + print(f" {'─'*6} {'─'*8} {'─'*7} {'─'*8} {'─'*44}") + best = None + for step, name, rid, passed, failed, total, rate, status in rows: + marker = "" + if best is None or rate > best[2]: + best = (step, name, rate, rid) + note = "" if status == "completed" else f" [{status}]" + print(f" {step:>6} {passed:>8} {total:>7} {rate:>7.1f}% {rid}{note}") + + # Trend diagnosis — the "early signal". + first_rate = rows[0][6] + last_step, last_name, _, _, _, _, last_rate, _ = rows[-1] + print(f"\n First eval (step {rows[0][0]}): {first_rate:.1f}% pass") + print(f" Latest eval (step {last_step}): {last_rate:.1f}% pass") + if best: + print(f" Best: {best[1]} — {best[2]:.1f}% pass ({best[3]})") + + delta = last_rate - first_rate + if len(rows) < 2: + print("\n ⏳ Only one eval so far — need at least two to read a trend.") + elif delta > 5: + trailing = last_rate - rows[max(0, len(rows) - 4)][6] + if trailing <= 0.5: + print(f"\n ⚠️ Improving overall (+{delta:.1f}%) but PLATEAUING in the last " + f"few evals. Consider whether more steps will help.") + else: + print(f"\n ✅ Learning: pass rate up +{delta:.1f}% since the first eval. " + f"Good early signal.") + elif delta < -5: + print(f"\n ⚠️ REGRESSING: pass rate down {delta:.1f}% since the first eval. " + f"Possible over-training or reward hacking — inspect a recent run's " + f"output_items with --run-id.") + else: + print(f"\n ⚠️ FLAT ({delta:+.1f}%): little movement. The grader may be " + f"miscalibrated, the task too hard, or hyperparameters off. Inspect " + f"per-sample results with --run-id, and see references/grader-design.md.") + + print("\n Inspect any step's per-sample results:") + print(f" python analyze_auto_evals.py --job-id {job_id} " + f"--run-id {best[3] if best else ''} --dump results.jsonl") + + +def dump_items(client, job_id, run_id, out_path): + """Write every per-sample output_item for one run to JSONL, or preview the first few.""" + job = client.fine_tuning.jobs.retrieve(job_id) + eval_id = _get_eval_id(job) + if not eval_id: + print("No eval attached to this job.") + return + + # Run-level pass/fail totals are already summarized on the run object — no need to + # scan every sample just to count them. + run = client.evals.runs.retrieve(run_id, eval_id=eval_id) + rc = run.result_counts + total = rc.total or 0 + rate = (rc.passed / total * 100) if total else 0.0 + print(f"\n {run.name} ({run_id}): {rc.passed}/{total} passed ({rate:.1f}%)") + + # Preview only: show the first 10 samples and stop (the API is slow per item). + if not out_path: + print("\n First samples (use --dump to write ALL to JSONL):") + n = 0 + for item in client.evals.runs.output_items.list(run_id, eval_id=eval_id, limit=10): + n += 1 + rec = item.model_dump() + res = (rec.get("results") or [{}])[0] + print(f" {rec['id']} passed={res.get('passed')} score={res.get('score')}") + if n >= 10: + break + return + + # Full dump: iterate all samples. NOTE: output_items carry the full sample (input, + # model output, tool calls), so the API is slow and *times out* at large page sizes + # (limit=100 → server-side 120s+ hang). Keep the page size small; the SDK + # auto-paginates as we iterate the cursor. + out = open(out_path, "w", encoding="utf-8") + n = passed = 0 + scores = [] + for item in client.evals.runs.output_items.list(run_id, eval_id=eval_id, limit=20): + n += 1 + rec = item.model_dump() + res = (rec.get("results") or [{}])[0] + if res.get("passed"): + passed += 1 + if res.get("score") is not None: + scores.append(res["score"]) + out.write(json.dumps(rec) + "\n") + if n % 20 == 0: + print(f" ...{n} items", flush=True) + out.close() + avg = sum(scores) / len(scores) if scores else 0.0 + print(f"\n Wrote {n} output items ({passed} passed" + + (f", avg score {avg:.3f}" if scores else "") + f") to {out_path}") + + +def main(): + parser = HelpOnErrorParser(description="Analyze the auto-generated evals on a fine-tuning job") + parser.add_argument("--base-url", default=os.environ.get("OPENAI_BASE_URL"), + help="Project /v1/ URL (preferred)") + parser.add_argument("--endpoint", default=os.environ.get("AZURE_OPENAI_ENDPOINT"), + help="Azure OpenAI endpoint (fallback)") + parser.add_argument("--project-endpoint", default=os.environ.get("AZURE_AI_PROJECT_ENDPOINT"), + help="Azure AI project endpoint (Foundry SDK)") + parser.add_argument("--api-key", default=os.environ.get("AZURE_OPENAI_API_KEY")) + parser.add_argument("--job-id", required=True, help="Fine-tuning job ID") + parser.add_argument("--run-id", help="Dump per-sample output_items for this eval run") + parser.add_argument("--dump", help="Write output_items to this JSONL path (with --run-id)") + args = parser.parse_args() + + client, _ = get_clients( + base_url=args.base_url, azure_endpoint=args.endpoint, + project_endpoint=args.project_endpoint, api_key=args.api_key, + ) + + if args.run_id: + dump_items(client, args.job_id, args.run_id, args.dump) + else: + curve(client, args.job_id) + + +if __name__ == "__main__": + main() From 99bb21e2baf86e0fd88be7dc67b65de9a3a73439 Mon Sep 17 00:00:00 2001 From: Alicia Frame Date: Tue, 7 Jul 2026 15:46:51 -0400 Subject: [PATCH 2/2] Remove unused variable in analyze_auto_evals curve loop Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- Skills/scripts/analyze_auto_evals.py | 1 - 1 file changed, 1 deletion(-) diff --git a/Skills/scripts/analyze_auto_evals.py b/Skills/scripts/analyze_auto_evals.py index b8d9022..4a7255f 100644 --- a/Skills/scripts/analyze_auto_evals.py +++ b/Skills/scripts/analyze_auto_evals.py @@ -88,7 +88,6 @@ def curve(client, job_id): print(f" {'─'*6} {'─'*8} {'─'*7} {'─'*8} {'─'*44}") best = None for step, name, rid, passed, failed, total, rate, status in rows: - marker = "" if best is None or rate > best[2]: best = (step, name, rate, rid) note = "" if status == "completed" else f" [{status}]"