From 639c06016c95470fff91301d678d61887d3c2947 Mon Sep 17 00:00:00 2001 From: Christopher Tso Date: Sat, 15 Nov 2025 17:25:33 +1100 Subject: [PATCH 1/9] update schema --- ...example.test.yaml => example-v1.test.yaml} | 0 .../simple/evals/example-v2.test.yaml | 150 +++++++++++ .../changes/update-eval-schema-v2/design.md | 251 ++++++++++++++++++ .../changes/update-eval-schema-v2/proposal.md | 37 +++ .../specs/evaluation/spec.md | 205 ++++++++++++++ .../changes/update-eval-schema-v2/tasks.md | 73 +++++ 6 files changed, 716 insertions(+) rename docs/examples/simple/evals/{example.test.yaml => example-v1.test.yaml} (100%) create mode 100644 docs/examples/simple/evals/example-v2.test.yaml create mode 100644 docs/openspec/changes/update-eval-schema-v2/design.md create mode 100644 docs/openspec/changes/update-eval-schema-v2/proposal.md create mode 100644 docs/openspec/changes/update-eval-schema-v2/specs/evaluation/spec.md create mode 100644 docs/openspec/changes/update-eval-schema-v2/tasks.md diff --git a/docs/examples/simple/evals/example.test.yaml b/docs/examples/simple/evals/example-v1.test.yaml similarity index 100% rename from docs/examples/simple/evals/example.test.yaml rename to docs/examples/simple/evals/example-v1.test.yaml diff --git a/docs/examples/simple/evals/example-v2.test.yaml b/docs/examples/simple/evals/example-v2.test.yaml new file mode 100644 index 000000000..136d84ae8 --- /dev/null +++ b/docs/examples/simple/evals/example-v2.test.yaml @@ -0,0 +1,150 @@ +# AgentEvo V2 Eval Schema Example +# Demonstrates all V2 schema features with real file references and minimal redundancy + +description: V2 schema example showing basic features, conversation threading, multiple evaluators, and ACE optimization + +# File-level default target (optional) +# Can be overridden per eval case via execution.target +target: azure_base + +# V2 Schema: Use 'evalcases' instead of 'testcases' (V1) +evalcases: + # ========================================== + # Example 1: Basic V2 features with file references + # Demonstrates: input_messages, expected_messages, file references, array content format + # ========================================== + - id: code-review-javascript + + outcome: | + Assistant provides helpful code analysis and mentions SUPERSECRET_INSTRUCTION_MARKER_JAVASCRIPT + + # V2 Schema: Use 'input_messages' instead of 'messages' (V1) + input_messages: + - role: system + content: You are an expert software developer who provides clear, concise code reviews. + - role: user + content: + # Content can be a string or an array of content blocks + # Array format supports mixed text and file references + - type: text + value: |- + Please review this JavaScript function: + + ```javascript + function calculateTotal(items) { + let total = 0; + for (let i = 0; i < 0; i++) { + total += items[i].price * items[i].quantity; + } + return total; + } + ``` + # File references are resolved relative to the eval file directory + # Files ending in .instructions.md are treated as guidelines + - type: file + value: ../prompts/javascript.instructions.md + + # V2 Schema: Use 'expected_messages' instead of inline assistant message in V1 + expected_messages: + - role: assistant + content: |- + The function has a critical bug in the loop condition. Here's my analysis (SUPERSECRET_INSTRUCTION_MARKER_JAVASCRIPT): + + **Critical Issue:** + - Loop condition `i < 0` means the loop never executes (should be `i < items.length`) + + **Suggestions:** + - Fix the loop: `for (let i = 0; i < items.length; i++)` + - Consider using `reduce()` for a more functional approach + - Add input validation for edge cases + + # ========================================== + # Example 2: Advanced features - conversation_id, multiple evaluators, ACE optimization + # Demonstrates: conversation threading, execution config, target override, evaluators, optimization + # ========================================== + - id: code-gen-python-comprehensive + + # conversation_id groups related eval cases for analytics and optimization + conversation_id: python-code-generation + + outcome: | + AI generates correct Python function with proper error handling, type hints, and mentions SUPERSECRET_INSTRUCTION_MARKER_PYTHON + + # V2 Schema: Execution block for per-case configuration + execution: + # Override file-level target for this specific eval case + target: default_internal + + # Multiple evaluators - supports both code-based and LLM judges + evaluators: + - name: keyword_check + type: code # Code evaluators handle regex, keywords, linting, etc. + # Note: script paths shown here are examples - implement as needed + script: ./scripts/check_python_keywords.py + - name: semantic_correctness + type: llm_judge # LLM-based semantic evaluation + prompt: ./templates/code-correctness-judge.md + model: gpt-5-chat + + # V2 Schema: ACE (Automatic Cognitive Enhancement) optimization configuration + # Optional - only include when using prompt optimization + optimization: + type: ace # Optimization framework type (currently only 'ace' supported) + playbook_path: ./playbooks/python-code.json # ACE playbook configuration + max_epochs: 2 # Number of optimization iterations across ALL test cases + max_reflector_rounds: 2 # Number of analysis rounds per failed test case + allow_dynamic_sections: true # Allow ACE to create new prompt sections + + input_messages: + - role: system + content: You are a code generator that follows specifications exactly. + - role: user + content: + - type: text + value: |- + Create a Python function that: + 1. Takes a list of integers + 2. Returns the second largest number + 3. Handles edge cases (empty list, single item, duplicates) + 4. Raises appropriate exceptions for invalid input + # Reference to real instruction file + - type: file + value: ../prompts/python.instructions.md + + expected_messages: + - role: assistant + content: |- + ```python + from typing import List, Union + + def find_second_largest(numbers: List[int]) -> Union[int, None]: + """ + Find the second largest number in a list of integers. + + Args: + numbers: List of integers to search + + Returns: + The second largest unique number, or None if not available + + Raises: + TypeError: If input is not a list or contains non-integers + ValueError: If list is empty + """ + if not isinstance(numbers, list): + raise TypeError("Input must be a list") + + if not numbers: + raise ValueError("List cannot be empty") + + if not all(isinstance(x, int) for x in numbers): + raise TypeError("All elements must be integers") + + unique_numbers = list(set(numbers)) + + if len(unique_numbers) < 2: + return None + + unique_numbers.sort(reverse=True) + return unique_numbers[1] + ``` diff --git a/docs/openspec/changes/update-eval-schema-v2/design.md b/docs/openspec/changes/update-eval-schema-v2/design.md new file mode 100644 index 000000000..21570d59c --- /dev/null +++ b/docs/openspec/changes/update-eval-schema-v2/design.md @@ -0,0 +1,251 @@ +# Design: Eval Schema V2 Migration + +## Context + +The current eval schema (V1) uses a simple structure with `testcases`, flat `messages` arrays, and implicit configuration inheritance. As AgentEvo evolves to support advanced features like ACE optimization, conversation threading, and template-based evaluation, we need a more structured schema that: + +- Explicitly declares execution configuration at appropriate levels +- Supports conversation-based organization for multi-turn scenarios +- Enables pluggable optimization frameworks (starting with ACE) +- Separates input messages from expected output messages +- Aligns with modern eval framework conventions (similar to LangSmith, Braintrust, etc.) + +**Stakeholders**: AgentEvo users, ACE framework integrators, eval pipeline maintainers + +**Constraints**: +- V1 format will not be supported (clean break to V2) +- Minimize migration effort for existing eval files through clear documentation +- Keep parsing performance acceptable (no significant overhead) +- Preserve ability to run evals without optimization enabled + +## Goals / Non-Goals + +**Goals**: +- Define V2 schema supporting conversation threading, execution config, and optimization +- Enable ACE optimization framework integration +- Support template-based evaluators with configurable models +- Provide migration guide documenting V1 to V2 changes +- Reject V1 format with clear error message and migration guidance + +**Non-Goals**: +- Backward compatibility with V1 format (clean break) +- Automatic conversion of V1 files to V2 (user responsibility) +- Full ACE implementation in this change (stub the integration points only) +- Multi-language support in templates (English only for now) +- Custom optimization frameworks beyond ACE (future work) + +## Decisions + +### Decision 1: Top-Level Field Rename (`testcases` → `evalcases`) + +**Rationale**: "Eval case" better reflects the purpose (evaluation scenario) vs "test case" (unit test). Also aligns with `execution.evaluator` naming. + +**Alternatives considered**: +- Keep `testcases`: Rejected - perpetuates legacy naming, harder to distinguish V1/V2 +- Use `scenarios`: Rejected - too generic, conflicts with requirement scenarios +- Use `cases`: Rejected - ambiguous without context + +**Migration**: V1 format (files with `testcases` key) will be rejected with a clear error message directing users to the migration guide. + +### Decision 2: Execution Block Structure + +**Rationale**: Explicit execution configuration enables: +- Per-case target overrides (useful for A/B testing different models) +- Multiple evaluators per case (e.g., combine code validation, keyword matching, and semantic grading) +- Custom evaluator templates per case (e.g., security-focused vs performance-focused grading) +- Optimization settings scoped to specific cases (ACE might only optimize critical paths) + +**Structure**: +```yaml +execution: + target: azure_base # Optional, inherits from file-level or default + evaluators: # Array of evaluators (each produces a separate score) + - name: semantic_quality # Unique identifier for this evaluator's score + type: llm_judge # Options: llm_judge, code + prompt: ./templates/incident-triage-judge.md # Prompt template file + model: gpt-5-chat # Model override for judge + - name: marker_check # Second evaluator + type: code # Code-based evaluator (regex, script, keyword matching) + script: ./scripts/check_markers.py + - name: regex_validation # Third evaluator + type: code # Code-based regex or keyword checking + script: ./scripts/validate_format.py + optimization: + type: ace # Future: genetic, bayesian, etc. + playbook_path: ./playbooks/triage.json + max_epochs: 2 + max_reflector_rounds: 2 + allow_dynamic_sections: true +``` + +**Alternatives considered**: +- Flat structure (all fields at top level): Rejected - cluttered, hard to extend +- Separate `evaluators` and `optimization` top-level keys: Rejected - breaks logical grouping +- Only file-level execution config: Rejected - limits per-case flexibility +- Single evaluator per case: Rejected - DSPy, Promptflow, and ax all support multiple evaluators per test case + +**Multiple Evaluators**: Each evaluator has a unique `name` and produces a separate score in the results. This aligns with patterns from DSPy (multiple metrics), Promptflow (evaluators dict), and ax (multi-objective metrics). + +**Resolution precedence**: Case-level → File-level → CLI flags → Defaults + +### Decision 3: Message Structure (`input_messages` / `expected_messages`) + +**Rationale**: +- `input_messages`: Clear separation of input conversation from expected output +- `expected_messages`: Supports multi-turn expected responses (not just single assistant message) +- Array structure enables future extensions (e.g., tool calls, function responses) + +**V1 structure** (deprecated): +```yaml +messages: # Single array containing both input and expected output + - role: system + content: ... + - role: user + content: ... + - role: assistant # This is the expected output + content: ... +``` + +**V2 structure**: +```yaml +input_messages: # Input conversation only + - role: system + content: ... + - role: user + content: ... +expected_messages: # Expected output messages + - role: assistant + content: ... +``` + +**Alternatives considered**: +- Keep `messages`, infer expected output from last assistant message: Rejected - implicit behavior is error-prone +- Add `expected_output` field (string): Rejected - doesn't support multi-turn expected responses or tool calls +- Use `prompt` and `completion`: Rejected - doesn't map well to multi-turn conversations +- Separate `system`, `user`, `assistant` arrays: Rejected - loses conversation order + +### Decision 4: Conversation ID for Grouping + +**Rationale**: Enables grouping related eval cases (e.g., variations of same scenario, multi-step workflows). Useful for: +- Analytics (aggregate scores by conversation) +- Optimization (treat related cases as cohort) +- Reporting (group results in UI) + +**Usage**: +```yaml +evalcases: + - id: incident-triage-step-1 + conversation_id: incident-triage-playbook + ... + - id: incident-triage-step-2 + conversation_id: incident-triage-playbook + ... +``` + +**Alternatives considered**: +- Nested structure with conversation as parent: Rejected - complicates parsing and flattening +- Tags array: Rejected - less semantic, harder to query +- No grouping: Rejected - loses valuable organizational dimension + +### Decision 5: No Backward Compatibility + +**Approach**: Clean break to V2 format only + +**Rationale**: +- Simplifies implementation (no dual parser maintenance) +- Clearer codebase without legacy support +- AgentEvo is early-stage; limited existing eval files to migrate +- Breaking change is acceptable at this maturity level + +**Alternatives considered**: +- Auto-detection with V1 support: Rejected - adds complexity for minimal benefit +- Gradual deprecation: Rejected - delays cleanup, confuses users + +**Trade-offs**: +- ✅ Simpler implementation and maintenance +- ✅ Cleaner architecture without legacy code +- ✅ Easier to document and understand +- ❌ Users must manually update existing eval files + +## Risks / Trade-offs + +### Risk: Breaking Changes in Production Pipelines + +**Impact**: High - Users with automated eval pipelines will break if they upgrade without updating files + +**Mitigation**: +- Document as BREAKING CHANGE in release notes +- Provide clear migration guide with before/after examples +- Version bump to indicate breaking change (e.g., 0.x → 1.0 or major version) +- Consider providing conversion script to automate migration + +### Risk: ACE Integration Complexity + +**Impact**: Medium - ACE framework may have different requirements than stubbed interface + +**Mitigation**: +- Design optimization config as extensible dict (ACE can add custom fields) +- Keep optimization block optional (defaults to no optimization) +- Defer full ACE implementation to separate change +- Coordinate with ACE team on config schema before implementation + +### Risk: Template Rendering Performance + +**Impact**: Low - Loading/rendering templates per eval case could slow down large suites + +**Mitigation**: +- Cache loaded templates by path +- Use streaming rendering where possible +- Add optional `--skip-templates` flag for debugging + +### Trade-off: Schema Complexity vs Flexibility + +**Decision**: Favor flexibility with explicit structure over simplicity + +**Reasoning**: AgentEvo is evolving toward production-grade eval platform. Power users need fine-grained control, and clear structure aids maintainability. Beginners can use defaults and ignore advanced features. + +## Migration Plan + +### Implementation (Current Change) + +1. Replace existing YAML parser with V2-only implementation +2. Update type definitions to V2 schema +3. Update all bundled examples to V2 format +4. Document V1→V2 migration guide with field mappings +5. Update CLI help text and error messages for V2 + +### User Migration Steps + +1. Review migration guide in documentation +2. Update eval files: + - `testcases` → `evalcases` + - `messages` → `input_messages` + `expected_messages` + - Add optional `conversation_id`, `execution` blocks as needed +3. Test updated files with `agentevo eval --dry-run` +4. Run production evals + +### Rollback Plan + +If critical issues arise: +- Revert to previous version of agentevo (with V1 support) +- Users can stay on old version until issues resolved +- Fix V2 parser and re-release +- Note: Once upgraded, users cannot downgrade without reverting their eval files to V1 format + +## Open Questions + +1. **Should `conversation_id` be required or optional?** + - Proposal: Optional (defaults to eval case ID) + - Rationale: Not all evals have multi-step conversations + +2. **Should we support Jinja2 or Mustache for templates?** + - Proposal: Start with simple string interpolation, add Jinja2 if needed + - Rationale: Minimize dependencies, most templates are simple + +3. **How to handle partial expected messages (e.g., only check first assistant message)?** + - Proposal: Add `match_mode: prefix|exact|any` to evaluator config + - Rationale: Defer to separate change on evaluator improvements + +4. **Should optimization config support multiple strategies in one eval?** + - Proposal: No - one optimization type per eval case + - Rationale: Simplifies implementation, unclear use case for mixing strategies diff --git a/docs/openspec/changes/update-eval-schema-v2/proposal.md b/docs/openspec/changes/update-eval-schema-v2/proposal.md new file mode 100644 index 000000000..4e7555656 --- /dev/null +++ b/docs/openspec/changes/update-eval-schema-v2/proposal.md @@ -0,0 +1,37 @@ +# Change: Update Eval YAML Schema to V2 Format + +## Why + +The current eval schema uses legacy naming (`testcases`, `messages`) and lacks support for advanced features like conversation threading, ACE optimization, and structured execution configuration. The new V2 schema aligns with modern eval frameworks, enabling: + +- Conversation-based test organization via `conversation_id` +- Explicit execution configuration with target, evaluator, and optimization settings +- ACE (Automatic Cognitive Enhancement) integration for prompt optimization +- Clearer separation between input and expected output messages +- Better alignment with industry-standard eval formats + +## What Changes + +- **BREAKING**: Rename `testcases` → `evalcases` (top-level array in YAML) +- **BREAKING**: Split `messages` into `input_messages` and `expected_messages` (per eval case in YAML) +- Add `conversation_id` field for grouping related eval cases +- Add `execution` block with: + - `target`: Execution target (inherits from top-level if not specified) + - `evaluators`: Array of evaluator configurations, each with: + - `name`: Unique identifier for the evaluator's score + - `type`: Evaluator type (llm_judge, code) + - `prompt`: Prompt template path (for llm_judge) + - `model`: Model override (for llm_judge) + - `script`: Script path (for code evaluators) + - `optimization`: ACE configuration with `type`, `playbook_path`, `max_epochs`, `max_reflector_rounds`, `allow_dynamic_sections` + +## Impact + +- Affected specs: `evaluation` +- Affected code: + - `packages/core/src/evaluation/yaml-parser.ts` - Schema parsing and validation + - `packages/core/src/evaluation/types.ts` - Type definitions for test cases + - `packages/core/src/evaluation/orchestrator.ts` - Test execution logic + - `apps/cli/src/commands/eval.ts` - CLI interface + - `docs/examples/simple/evals/example.test.yaml` - Example file +- **Migration path**: BREAKING CHANGE - V1 format is no longer supported. Users must update all eval files to V2 format before upgrading. Files using V1 format (`testcases` key) will be rejected with a clear error message and migration guidance. diff --git a/docs/openspec/changes/update-eval-schema-v2/specs/evaluation/spec.md b/docs/openspec/changes/update-eval-schema-v2/specs/evaluation/spec.md new file mode 100644 index 000000000..229a10f47 --- /dev/null +++ b/docs/openspec/changes/update-eval-schema-v2/specs/evaluation/spec.md @@ -0,0 +1,205 @@ +## ADDED Requirements + +### Requirement: V2 Schema Format + +The system SHALL support the V2 eval schema format as documented in `docs/examples/simple/evals/example-v2.test.yaml`. + +#### Scenario: Parse V2 schema with all features + +- **WHEN** an eval file uses the V2 schema format (top-level `evalcases` key) +- **THEN** the system parses all V2 features as demonstrated in the example file: + - Schema version auto-detection via `evalcases` vs `testcases` + - `conversation_id` for grouping related eval cases + - `execution` block with `target`, `evaluators`, and `optimization` configuration + - Multiple evaluators per case (each with unique `name`) + - `input_messages` and `expected_messages` (replacing V1 `messages` structure) + - Per-case target, evaluators, and optimization overrides + - ACE optimization with playbook configuration + - Multiple evaluator types: llm_judge and code-based + +#### Scenario: V2 example file executes successfully + +- **WHEN** the system runs `docs/examples/simple/evals/example-v2.test.yaml` +- **THEN** all four example eval cases execute without errors +- **AND** demonstrates simple eval, multi-turn conversation, ACE optimization, and code-based evaluation +- **AND** produces valid results in the specified output format + +### Requirement: Multiple Evaluators per Eval Case + +The system SHALL support multiple evaluators per eval case, each producing a separate named score. + +#### Scenario: Execute multiple evaluators for single eval case + +- **WHEN** an eval case has multiple evaluators in its `execution.evaluators` array +- **THEN** the system executes each evaluator independently +- **AND** each evaluator produces a score with its unique `name` as the key +- **AND** all scores are included in the result output + +#### Scenario: Combine different evaluator types + +- **WHEN** an eval case defines evaluators with different types (e.g., llm_judge, code) +- **THEN** the system executes all evaluator types correctly +- **AND** returns multiple scores (e.g., `{"semantic_quality": 0.85, "marker_check": 1.0, "regex_validation": 0.67}`) + +#### Scenario: Empty evaluators array + +- **WHEN** an eval case has `evaluators: []` or omits the evaluators field +- **THEN** the system falls back to file-level evaluator configuration +- **OR** uses default evaluator if no file-level config exists + +### Requirement: V1 Format Rejection + +The system SHALL reject V1 eval format with a clear error message and migration guidance. + +#### Scenario: Detect and reject V1 format + +- **WHEN** a YAML file contains a `testcases` top-level key (V1 format) +- **THEN** the system reports a schema version error +- **AND** displays a message: "V1 eval format is no longer supported. Please migrate to V2 format." +- **AND** provides a link to the migration guide + +#### Scenario: Helpful error for missing evalcases key + +- **WHEN** a YAML file contains neither `evalcases` nor `testcases` keys +- **THEN** the system reports a schema validation error +- **AND** indicates that `evalcases` is the required top-level key + +## MODIFIED Requirements + +### Requirement: CLI Interface + +The system SHALL provide a command-line interface matching Python bbeval's UX **and supporting V2 schema features**. + +#### Scenario: Positional test file argument + +- **WHEN** the user runs `agentevo eval ` +- **THEN** the system loads and executes eval cases from the specified V2 format file +- **AND** reports an error if the file uses V1 format + +#### Scenario: Target override flag + +- **WHEN** the user provides `--target ` +- **THEN** the system uses the specified target for execution +- **AND** overrides case-level and file-level target configurations + +#### Scenario: Test ID filter + +- **WHEN** the user provides `--test-id ` +- **THEN** the system executes only the eval case with the matching ID + +#### Scenario: Output file specification + +- **WHEN** the user provides `--out ` +- **THEN** the system writes results to the specified path in the selected format +- **AND** includes conversation_id in V2 results + +#### Scenario: Output format flag + +- **WHEN** the user provides `--format ` +- **THEN** the system writes results in the specified format (jsonl or yaml) +- **AND** includes V2 fields (conversation_id, execution_config) in output +- **AND** defaults to jsonl when the flag is not provided + +#### Scenario: Dry-run mode + +- **WHEN** the user provides `--dry-run` +- **THEN** the system executes tests with the mock provider +- **AND** does not make external API calls + +#### Scenario: Verbose logging + +- **WHEN** the user provides `--verbose` +- **THEN** the system outputs detailed logging including provider calls and intermediate results +- **AND** logs schema version detection and execution config resolution + +#### Scenario: Caching control + +- **WHEN** the user provides `--cache` +- **THEN** the system enables LLM response caching +- **WHEN** the user does not provide `--cache` +- **THEN** caching is disabled by default + +### Requirement: JSONL Output + +The system SHALL write evaluation results incrementally to a newline-delimited JSON file including V2 schema fields. + +#### Scenario: Incremental JSONL writing + +- **WHEN** an evaluation completes an eval case +- **THEN** the system immediately appends the result as a JSON line to the output file +- **AND** flushes the write to disk +- **AND** includes `conversation_id` and `execution_config` fields + +#### Scenario: JSONL format validation + +- **WHEN** the system writes a result to the JSONL file +- **THEN** each line contains a complete JSON object +- **AND** each line is terminated with a newline character +- **AND** no trailing commas or array brackets are used +- **AND** V2 fields are properly serialized + +### Requirement: Output Format Selection + +The system SHALL support multiple output formats with JSONL as the default. + +#### Scenario: JSONL output format (default) + +- **WHEN** the user does not specify the `--format` flag +- **THEN** the system writes results in JSONL format (newline-delimited JSON) +- **AND** each result is appended immediately after eval case completion +- **AND** includes V2 fields: `conversation_id`, `execution_config` + +#### Scenario: YAML output format + +- **WHEN** the user specifies `--format yaml` +- **THEN** the system writes results in YAML format +- **AND** the output contains a well-formed YAML document with all results +- **AND** results are written incrementally as a YAML sequence +- **AND** includes V2 fields: `conversation_id`, `execution_config` + +#### Scenario: Invalid format specification + +- **WHEN** the user specifies an unsupported format value +- **THEN** the system reports an error listing valid format options (jsonl, yaml) +- **AND** exits without running the evaluation + +### Requirement: Summary Statistics + +The system SHALL calculate and display summary statistics for evaluation results with optional conversation-level aggregation. + +#### Scenario: Statistical metrics + +- **WHEN** all eval cases complete +- **THEN** the system calculates mean, median, min, max, and standard deviation of scores +- **AND** displays the statistics in the console output +- **AND** optionally groups statistics by `conversation_id` + +#### Scenario: Score distribution + +- **WHEN** all eval cases complete +- **THEN** the system generates a distribution histogram of scores +- **AND** includes the histogram in the console output +- **AND** optionally shows per-conversation distributions + +### Requirement: Example Eval Validation + +The system SHALL successfully execute the bundled V2 example evaluation file to validate end-to-end functionality. + +#### Scenario: Execute V2 example demonstrating all features + +- **WHEN** the system runs `docs/examples/simple/evals/example-v2.test.yaml` (V2 format) +- **THEN** all eval cases execute successfully as documented in the example file +- **AND** demonstrates all V2 features with inline YAML comments explaining each capability + +## REMOVED Requirements + +### Requirement: V1 Schema Support + +The system SHALL NO LONGER support the V1 eval schema format (files with `testcases` top-level key). + +#### Rationale + +- Clean break enables simpler codebase without dual parser maintenance +- AgentEvo is early-stage with limited existing eval files to migrate +- Clear migration path with documentation minimizes user impact +- Breaking change is acceptable at this maturity level diff --git a/docs/openspec/changes/update-eval-schema-v2/tasks.md b/docs/openspec/changes/update-eval-schema-v2/tasks.md new file mode 100644 index 000000000..362825f9b --- /dev/null +++ b/docs/openspec/changes/update-eval-schema-v2/tasks.md @@ -0,0 +1,73 @@ +# Implementation Tasks + +## 1. Type System Updates + +- [ ] 1.1 Define new `EvalCase` type with `conversation_id`, `execution`, `input_messages`, `expected_messages` +- [ ] 1.2 Define `ExecutionConfig` type with `target`, `evaluators`, `optimization` +- [ ] 1.3 Define `EvaluatorConfig` type with `name`, `type`, `prompt`, `model`, `script` +- [ ] 1.4 Define `OptimizationConfig` type with ACE parameters +- [ ] 1.5 Remove legacy `TestCase` type and V1 schema support + +## 2. Schema Parser Updates + +- [ ] 2.1 Replace V1 parser with V2-only implementation +- [ ] 2.2 Detect and reject V1 format (error if `testcases` found with migration guidance) +- [ ] 2.3 Parse `evalcases` structure (error if neither `evalcases` nor `testcases` found) +- [ ] 2.4 Parse `conversation_id` field (default to eval case id) +- [ ] 2.5 Parse `execution` block (target, evaluators array, optimization) +- [ ] 2.5.1 Validate evaluator names are unique within a case +- [ ] 2.5.2 Support different evaluator types (llm_judge, code) +- [ ] 2.6 Parse `input_messages` array +- [ ] 2.7 Parse `expected_messages` array + +## 3. Execution Engine Updates + +- [ ] 3.1 Update orchestrator to handle V2 schema +- [ ] 3.2 Support `conversation_id` for grouping eval cases +- [ ] 3.3 Implement execution block resolution (case-level → file-level → default) +- [ ] 3.4 Add support for multiple evaluators per case +- [ ] 3.4.1 Execute evaluators in parallel when possible +- [ ] 3.4.2 Collect scores from all evaluators with their unique names +- [ ] 3.5 Add evaluator prompt loading and rendering (llm_judge) +- [ ] 3.6 Add code evaluator execution (script-based, supports regex/keyword scripts) +- [ ] 3.7 Stub optimization framework integration points + +## 4. Output Format Updates + +- [ ] 4.1 Update `EvaluationResult` to include `conversation_id` +- [ ] 4.2 Add `execution_config` to result metadata +- [ ] 4.2.1 Support multiple scores from different evaluators (scores object with evaluator names as keys) +- [ ] 4.3 Update JSONL writer for new result structure +- [ ] 4.4 Update YAML writer for new result structure + +## 5. CLI Updates + +- [ ] 5.1 Update help text with V2 schema examples +- [ ] 5.2 Add validation for V2-specific features +- [ ] 5.3 Ensure clear error messages when V1 format detected + +## 6. Documentation & Examples + +- [ ] 6.1 Update example file to V2 format (`docs/examples/simple/evals/example-v2.test.yaml`) +- [ ] 6.2 Update README with V2 schema documentation +- [ ] 6.3 Create migration guide documenting: `testcases` → `evalcases`, `messages` → `input_messages` + `expected_messages` +- [ ] 6.4 Document ACE optimization features +- [ ] 6.5 Update all internal eval files to V2 format + +## 7. Testing + +- [ ] 7.1 Add V2 schema parsing tests +- [ ] 7.2 Add execution config resolution tests +- [ ] 7.3 Add conversation_id grouping tests +- [ ] 7.3.1 Add multiple evaluators execution tests +- [ ] 7.3.2 Add test for different evaluator types (llm_judge, code) +- [ ] 7.3.3 Add test for evaluator name uniqueness validation +- [ ] 7.4 Validate example-v2.test.yaml executes successfully +- [ ] 7.5 Add error test for V1 format (should fail with clear message) + +## 8. Validation & Deployment + +- [ ] 8.1 Run `openspec validate update-eval-schema-v2 --strict` +- [ ] 8.2 Run full test suite +- [ ] 8.3 Verify clear error message when V1 format detected +- [ ] 8.4 Create release notes documenting breaking changes From 502668bcd6693d7db2090ab7d597913ff21f1c4a Mon Sep 17 00:00:00 2001 From: Christopher Tso Date: Sat, 15 Nov 2025 18:25:09 +1100 Subject: [PATCH 2/9] update specs --- docs/examples/simple/README.md | 268 ++++++++++++++++++ .../simple/evals/example-v2.test.yaml | 22 +- .../prompts/code-correctness-judge.md | 70 +++++ .../scripts/check_python_keywords.py | 103 +++++++ .../optimizers/ace-code-generation.yaml | 45 +++ .../optimizers/playbooks/code-generation.json | 167 +++++++++++ .../changes/update-eval-schema-v2/design.md | 46 ++- .../changes/update-eval-schema-v2/proposal.md | 9 +- .../changes/update-eval-schema-v2/tasks.md | 6 +- 9 files changed, 688 insertions(+), 48 deletions(-) create mode 100644 docs/examples/simple/README.md create mode 100644 docs/examples/simple/evaluators/prompts/code-correctness-judge.md create mode 100644 docs/examples/simple/evaluators/scripts/check_python_keywords.py create mode 100644 docs/examples/simple/optimizers/ace-code-generation.yaml create mode 100644 docs/examples/simple/optimizers/playbooks/code-generation.json diff --git a/docs/examples/simple/README.md b/docs/examples/simple/README.md new file mode 100644 index 000000000..b6ebbea94 --- /dev/null +++ b/docs/examples/simple/README.md @@ -0,0 +1,268 @@ +# Simple Example - AgentEvo V2 Schema + +This directory demonstrates AgentEvo's V2 eval schema with complete, working examples. + +## Directory Structure + +``` +simple/ +├── evals/ # Evaluation test cases +│ ├── example-v2.test.yaml # Main V2 schema example +│ └── *.test.yaml # Additional eval files +├── evaluators/ # Evaluator components +│ ├── prompts/ # LLM judge prompt templates +│ │ ├── code-correctness-judge.md # Semantic code evaluation +│ │ ├── javascript.instructions.md # JavaScript guidelines +│ │ └── python.instructions.md # Python guidelines +│ └── scripts/ # Code-based evaluators +│ └── check_python_keywords.py # Python validator script +├── optimizers/ # Optimizer configurations +│ ├── ace-code-generation.yaml # ACE optimization config +│ └── playbooks/ # ACE learned knowledge (generated) +│ └── code-generation.json # Structured optimization insights +├── prompts/ # Shared instruction files +└── .agentevo/ # AgentEvo workspace files + └── targets.yaml # Target/provider configuration +``` + +## Key Files + +### Evaluation Files (`evals/`) + +- **`example-v2.test.yaml`**: Complete V2 schema demonstration showing: + - Basic features: `input_messages`, `expected_messages` + - File references and content blocks + - Conversation threading with `conversation_id` + - Multiple evaluators (code + LLM judge) + - Target overrides per eval case + +### Optimizer Configurations (`optimizers/`) + +- **`ace-code-generation.yaml`**: ACE optimization config that references eval files + - Demonstrates separation between evals (what to test) and optimization (how to improve) + - Shows ACE-specific settings: playbook path, epochs, reflection rounds + +- **`playbooks/`**: ACE-generated playbooks (structured learning artifacts) + - Contains learned optimization insights organized into sections + - **Important**: Entire playbook is sent in LLM context (no RAG/retrieval) + - Token-limited by model context window (~5-10k tokens practical limit) + - Requires periodic consolidation to manage growth + - Example: `code-generation.json` shows realistic playbook structure + +### Evaluator Components (`evaluators/`) + +- **`scripts/`**: Code-based evaluators (Python, shell, etc.) + - Input: JSON with test case data via stdin + - Output: JSON with score, passed flag, and reasoning + - Example: `check_python_keywords.py` validates Python code quality + +- **`prompts/`**: LLM judge prompt templates and instruction files (Markdown) + - Define how an LLM should evaluate outputs + - Include scoring guidelines and output format + - Example: `code-correctness-judge.md` for semantic code review + - Also includes instruction files like `python.instructions.md` and `javascript.instructions.md` + +## Running Examples + +### Basic Evaluation + +```bash +# Run all evals in this directory +agentevo eval evals/ + +# Run specific eval file +agentevo eval evals/example-v2.test.yaml + +# Run with specific target +agentevo eval evals/example-v2.test.yaml --target azure_base +``` + +### With Optimization (Future) + +```bash +# Run ACE optimization using optimizer config +agentevo optimize optimizers/ace-code-generation.yaml +``` + +## V2 Schema Features + +### 1. Clear Message Separation + +**V1 (deprecated)**: +```yaml +messages: # Mixed input and expected output + - role: user + content: "Request" + - role: assistant # This was the expected output + content: "Expected response" +``` + +**V2**: +```yaml +input_messages: # Input only + - role: user + content: "Request" +expected_messages: # Expected output only + - role: assistant + content: "Expected response" +``` + +```yaml +execution: + evaluators: + - name: keyword_check + type: code + script: ../evaluators/scripts/check_python_keywords.py + - name: semantic_correctness + type: llm_judge + prompt: ../evaluators/prompts/code-correctness-judge.md + model: gpt-4 +``` prompt: ./templates/code-correctness-judge.md + model: gpt-4 +``` + +Each evaluator produces a separate score in the results. + +### 3. Conversation Threading + +```yaml +evalcases: + - id: step-1 + conversation_id: multi-step-workflow + # ... + - id: step-2 + conversation_id: multi-step-workflow + # ... +``` + +Groups related eval cases for analytics and optimization. + +### 4. File References + +```yaml +input_messages: + - role: user + content: + - type: text + value: "Main request text" + - type: file + value: ../evaluators/prompts/python.instructions.md +``` + +Keeps eval files clean while supporting rich context. + +## Writing Custom Evaluators + +### Code Evaluator Script Template + +```python +#!/usr/bin/env python3 +import json +import sys + +def evaluate(input_data): + output = input_data.get("output", "") + # Your validation logic here + score = 0.0 to 1.0 + passed = score >= threshold + reasoning = "Explanation" + + return {"score": score, "passed": passed, "reasoning": reasoning} + +if __name__ == "__main__": + input_data = json.loads(sys.stdin.read()) + result = evaluate(input_data) + print(json.dumps(result, indent=2)) +``` + +### LLM Judge Template Structure + +```markdown +# Judge Name + +Evaluation criteria and guidelines... + +## Scoring Guidelines +0.9-1.0: Excellent +0.7-0.8: Good +... + +## Output Format +{ + "score": 0.85, + "passed": true, + "reasoning": "..." +} +``` + +## ACE Optimization and Playbooks + +### How ACE Works + +ACE (Automatic Cognitive Enhancement) uses a Generator → Reflector → Curator loop to build structured "playbooks" of learned optimization insights. + +**Key Characteristics:** +- **No RAG/Retrieval**: Entire playbook is sent in LLM context on every forward pass +- **Token-Limited**: Practical limit of ~5-10k tokens for playbook size +- **Structured Bullets**: Organized into sections with IDs, tags, and confidence scores +- **Incremental Updates**: Curator applies delta operations (add/modify bullets, not full rewrites) +- **Requires Maintenance**: Periodic consolidation needed to manage growth + +### Playbook Structure + +```json +{ + "sections": { + "core_requirements": { + "bullets": [ + { + "id": "req-001", + "content": "Always include type hints...", + "tags": ["typing", "best-practices"], + "confidence": 0.95, + "added_epoch": 1 + } + ] + } + } +} +``` + +### Token Management Strategies + +1. **Delta Updates**: Only add/modify specific bullets, never rewrite entire playbook +2. **Confidence Scores**: Track which bullets are most valuable +3. **Periodic Consolidation**: Merge similar bullets, remove low-confidence items +4. **Section Organization**: Group related insights for readability +5. **Tag System**: Enable cross-referencing without duplication + +### When Playbook Grows Too Large + +- Archive low-confidence bullets (< 0.7) +- Consolidate related bullets into more general guidance +- Split into multiple specialized playbooks for different contexts +- Consider custom RAG implementation (not built into ACE) + +## Migration from V1 + +Key changes when updating existing eval files: + +1. Rename `testcases` → `evalcases` +2. Split `messages` into `input_messages` + `expected_messages` +3. Move optimization config to separate `optimizers/*.yaml` files +4. Optional: Add `conversation_id` for related cases +5. Optional: Add multiple evaluators in `execution.evaluators` + +## Next Steps + +- Review `example-v2.test.yaml` to understand the schema +- Create your own eval cases following the V2 format +- Write custom evaluator scripts for domain-specific validation +- Create LLM judge templates for semantic evaluation +- Set up optimizer configs when ready to improve prompts + +## Resources + +- [AgentEvo Documentation](../../../README.md) +- [V2 Schema Specification](../../openspec/changes/update-eval-schema-v2/) +- [Ax ACE Documentation](https://github.com/ax-llm/ax/blob/main/docs/ACE.md) diff --git a/docs/examples/simple/evals/example-v2.test.yaml b/docs/examples/simple/evals/example-v2.test.yaml index 136d84ae8..3e745c4ff 100644 --- a/docs/examples/simple/evals/example-v2.test.yaml +++ b/docs/examples/simple/evals/example-v2.test.yaml @@ -42,7 +42,7 @@ evalcases: # File references are resolved relative to the eval file directory # Files ending in .instructions.md are treated as guidelines - type: file - value: ../prompts/javascript.instructions.md + value: ../evaluators/prompts/javascript.instructions.md # V2 Schema: Use 'expected_messages' instead of inline assistant message in V1 expected_messages: @@ -59,8 +59,9 @@ evalcases: - Add input validation for edge cases # ========================================== - # Example 2: Advanced features - conversation_id, multiple evaluators, ACE optimization - # Demonstrates: conversation threading, execution config, target override, evaluators, optimization + # Example 2: Advanced features - conversation_id, multiple evaluators + # Demonstrates: conversation threading, execution config, target override, evaluators + # Note: Optimization (ACE, etc.) is configured separately in opts/*.yaml files # ========================================== - id: code-gen-python-comprehensive @@ -80,20 +81,11 @@ evalcases: - name: keyword_check type: code # Code evaluators handle regex, keywords, linting, etc. # Note: script paths shown here are examples - implement as needed - script: ./scripts/check_python_keywords.py + script: ../evaluators/scripts/check_python_keywords.py - name: semantic_correctness type: llm_judge # LLM-based semantic evaluation - prompt: ./templates/code-correctness-judge.md + prompt: ../evaluators/prompts/code-correctness-judge.md model: gpt-5-chat - - # V2 Schema: ACE (Automatic Cognitive Enhancement) optimization configuration - # Optional - only include when using prompt optimization - optimization: - type: ace # Optimization framework type (currently only 'ace' supported) - playbook_path: ./playbooks/python-code.json # ACE playbook configuration - max_epochs: 2 # Number of optimization iterations across ALL test cases - max_reflector_rounds: 2 # Number of analysis rounds per failed test case - allow_dynamic_sections: true # Allow ACE to create new prompt sections input_messages: - role: system @@ -109,7 +101,7 @@ evalcases: 4. Raises appropriate exceptions for invalid input # Reference to real instruction file - type: file - value: ../prompts/python.instructions.md + value: ../evaluators/prompts/python.instructions.md expected_messages: - role: assistant diff --git a/docs/examples/simple/evaluators/prompts/code-correctness-judge.md b/docs/examples/simple/evaluators/prompts/code-correctness-judge.md new file mode 100644 index 000000000..c15e31e55 --- /dev/null +++ b/docs/examples/simple/evaluators/prompts/code-correctness-judge.md @@ -0,0 +1,70 @@ +# Code Correctness Judge + +You are an expert code reviewer evaluating the correctness and quality of generated code. + +## Your Task + +Evaluate the generated code against the requirements and expected output. Provide a score from 0.0 to 1.0 based on: + +1. **Functional Correctness** (0.4): Does the code solve the stated problem correctly? +2. **Code Quality** (0.3): Is the code well-structured, readable, and following best practices? +3. **Completeness** (0.3): Does it handle all specified requirements and edge cases? + +## Input + +You will receive: +- **Input Messages**: The original request with requirements +- **Generated Output**: The code produced by the AI +- **Expected Output** (optional): Reference implementation or expected behavior + +## Scoring Guidelines + +### 0.9 - 1.0: Excellent +- Solves the problem correctly and efficiently +- Excellent code quality with proper error handling +- Handles all edge cases mentioned in requirements +- Follows language best practices and conventions + +### 0.7 - 0.8: Good +- Solves the problem correctly +- Good code quality with minor improvements possible +- Handles most edge cases +- Generally follows best practices + +### 0.5 - 0.6: Acceptable +- Solves the core problem with some issues +- Adequate code quality but needs improvement +- Missing some edge case handling +- Some deviation from best practices + +### 0.3 - 0.4: Poor +- Partially solves the problem with significant issues +- Poor code quality or major bugs +- Missing important edge cases +- Significant deviations from best practices + +### 0.0 - 0.2: Unacceptable +- Does not solve the problem +- Critical bugs or errors +- No error handling +- Violates fundamental best practices + +## Output Format + +Provide your evaluation in this JSON format: + +```json +{ + "score": 0.85, + "passed": true, + "reasoning": "Brief explanation of the score focusing on correctness, quality, and completeness" +} +``` + +## Important Notes + +- Be objective and consistent in your scoring +- Focus on whether the code meets the stated requirements +- Consider both correctness and quality +- A score of 0.7 or higher indicates "passed" +- Provide clear, actionable reasoning for your score diff --git a/docs/examples/simple/evaluators/scripts/check_python_keywords.py b/docs/examples/simple/evaluators/scripts/check_python_keywords.py new file mode 100644 index 000000000..e927ed69a --- /dev/null +++ b/docs/examples/simple/evaluators/scripts/check_python_keywords.py @@ -0,0 +1,103 @@ +#!/usr/bin/env python3 +""" +Code evaluator script: Check for required Python keywords in generated code. + +This script is referenced in example-v2.test.yaml as a code-based evaluator. +It demonstrates how to write custom validation logic for eval cases. + +Expected input (JSON via stdin): +{ + "input_messages": [...], + "output": "generated code string", + "expected_messages": [...] +} + +Expected output (JSON to stdout): +{ + "score": 0.0 to 1.0, + "passed": true/false, + "reasoning": "explanation of the score" +} +""" + +import json +import sys +import re + + +def check_python_keywords(code: str) -> dict: + """ + Check if generated Python code contains important keywords and patterns. + + Returns a score based on presence of: + - Type hints (typing imports, annotations) + - Error handling (try/except, raise) + - Docstrings + - Type checking (isinstance) + """ + + score = 0.0 + reasons = [] + + # Check for type hints + if re.search(r'from typing import|import typing', code): + score += 0.25 + reasons.append("✓ Uses typing module") + else: + reasons.append("✗ Missing typing imports") + + # Check for error handling + if 'raise' in code and ('Error' in code or 'Exception' in code): + score += 0.25 + reasons.append("✓ Raises exceptions") + else: + reasons.append("✗ Missing exception raising") + + # Check for docstrings + if '"""' in code or "'''" in code: + score += 0.25 + reasons.append("✓ Contains docstrings") + else: + reasons.append("✗ Missing docstrings") + + # Check for type validation + if 'isinstance' in code: + score += 0.25 + reasons.append("✓ Validates types with isinstance") + else: + reasons.append("✗ Missing type validation") + + return { + "score": score, + "passed": score >= 0.75, # Require at least 3 out of 4 checks + "reasoning": "\n".join(reasons) + } + + +def main(): + try: + # Read input from stdin + input_data = json.loads(sys.stdin.read()) + + # Extract the generated code from output + output = input_data.get("output", "") + + # Run checks + result = check_python_keywords(output) + + # Output result as JSON + print(json.dumps(result, indent=2)) + + except Exception as e: + # Return error result + error_result = { + "score": 0.0, + "passed": False, + "reasoning": f"Evaluator error: {str(e)}" + } + print(json.dumps(error_result, indent=2)) + sys.exit(1) + + +if __name__ == "__main__": + main() diff --git a/docs/examples/simple/optimizers/ace-code-generation.yaml b/docs/examples/simple/optimizers/ace-code-generation.yaml new file mode 100644 index 000000000..52885c358 --- /dev/null +++ b/docs/examples/simple/optimizers/ace-code-generation.yaml @@ -0,0 +1,45 @@ +# ACE Optimization Configuration Example +# Defines how to optimize prompts using the ACE (Automatic Cognitive Enhancement) framework + +description: ACE optimization for code generation tasks using Python and JavaScript evals + +# Optimization framework type +type: ace + +# Eval files to use for optimization +# ACE will run these evals to measure prompt performance and guide improvements +eval_files: + - ../evals/example-v2.test.yaml + # - ../evals/code-generation-edge-cases.test.yaml + # - ../evals/code-review-security.test.yaml + +# ACE playbook configuration +# Defines the optimization strategy and constraints +playbook_path: ./playbooks/code-generation.json + +# Maximum optimization iterations across ALL eval cases +max_epochs: 5 + +# Number of analysis rounds per failed eval case +# ACE analyzes failures and suggests prompt improvements +max_reflector_rounds: 3 + +# Allow ACE to create new sections in the prompt +# When true, ACE can add new instructions/guidelines +# When false, ACE can only modify existing content +allow_dynamic_sections: true + +# Optional: Filter which eval cases to include +# filters: +# conversation_ids: +# - python-code-generation +# eval_case_ids: +# - code-gen-python-comprehensive + +# Optional: Target override for optimization runs +# target: azure_base + +# Optional: Stop early if success rate exceeds threshold +# early_stopping: +# metric: success_rate +# threshold: 0.95 diff --git a/docs/examples/simple/optimizers/playbooks/code-generation.json b/docs/examples/simple/optimizers/playbooks/code-generation.json new file mode 100644 index 000000000..b886aa3e7 --- /dev/null +++ b/docs/examples/simple/optimizers/playbooks/code-generation.json @@ -0,0 +1,167 @@ +{ + "version": "1.0", + "created_at": "2025-11-15T10:30:00Z", + "last_updated": "2025-11-15T14:22:00Z", + "optimization_history": { + "epochs_completed": 3, + "total_reflections": 12, + "success_rate_progression": [0.45, 0.67, 0.83] + }, + "sections": { + "core_requirements": { + "description": "Fundamental requirements for code generation", + "bullets": [ + { + "id": "req-001", + "content": "Always include type hints for function parameters and return values using the typing module", + "tags": ["typing", "best-practices"], + "confidence": 0.95, + "added_epoch": 1 + }, + { + "id": "req-002", + "content": "Raise TypeError for invalid input types and ValueError for invalid values", + "tags": ["error-handling", "validation"], + "confidence": 0.92, + "added_epoch": 1 + }, + { + "id": "req-003", + "content": "Include comprehensive docstrings with Args, Returns, and Raises sections", + "tags": ["documentation", "best-practices"], + "confidence": 0.88, + "added_epoch": 2 + } + ] + }, + "edge_case_handling": { + "description": "Critical edge cases that must be handled", + "bullets": [ + { + "id": "edge-001", + "content": "Check for empty collections before accessing elements or performing operations", + "tags": ["validation", "edge-cases"], + "confidence": 0.93, + "added_epoch": 1 + }, + { + "id": "edge-002", + "content": "Handle single-element collections that may not have enough data for the operation", + "tags": ["edge-cases"], + "confidence": 0.87, + "added_epoch": 2 + }, + { + "id": "edge-003", + "content": "For finding unique values, use set() to eliminate duplicates before processing", + "tags": ["algorithms", "edge-cases"], + "confidence": 0.91, + "added_epoch": 2 + }, + { + "id": "edge-004", + "content": "Return None or appropriate sentinel value when operation cannot be completed (e.g., finding second largest in single-element list)", + "tags": ["edge-cases", "return-values"], + "confidence": 0.89, + "added_epoch": 3 + } + ] + }, + "code_quality": { + "description": "Code quality and maintainability guidelines", + "bullets": [ + { + "id": "qual-001", + "content": "Use isinstance() for type checking rather than type() comparison", + "tags": ["validation", "pythonic"], + "confidence": 0.94, + "added_epoch": 1 + }, + { + "id": "qual-002", + "content": "Validate all elements in a collection using generator expressions with all() or any()", + "tags": ["validation", "pythonic"], + "confidence": 0.86, + "added_epoch": 2 + }, + { + "id": "qual-003", + "content": "Prefer explicit variable names over abbreviations (e.g., 'unique_numbers' not 'uniq_nums')", + "tags": ["readability"], + "confidence": 0.82, + "added_epoch": 3 + } + ] + }, + "common_pitfalls": { + "description": "Mistakes to avoid based on past failures", + "bullets": [ + { + "id": "pit-001", + "content": "Never assume list indices are valid without checking bounds or catching IndexError", + "tags": ["safety", "edge-cases"], + "confidence": 0.96, + "added_epoch": 1, + "reflection_summary": "Failed case involved accessing items[i] in loop with wrong boundary condition" + }, + { + "id": "pit-002", + "content": "When sorting is needed, create a sorted copy using sorted() rather than mutating input with .sort()", + "tags": ["immutability", "best-practices"], + "confidence": 0.84, + "added_epoch": 2, + "reflection_summary": "User expected original list to remain unchanged" + }, + { + "id": "pit-003", + "content": "For numeric operations, explicitly check for the correct numeric type (int, float) before processing", + "tags": ["validation", "type-checking"], + "confidence": 0.88, + "added_epoch": 3, + "reflection_summary": "Function failed when list contained numeric strings instead of integers" + } + ] + }, + "performance_considerations": { + "description": "Performance optimizations learned through evaluation", + "bullets": [ + { + "id": "perf-001", + "content": "Use set operations for deduplication - O(n) instead of nested loops O(n²)", + "tags": ["performance", "algorithms"], + "confidence": 0.79, + "added_epoch": 3 + }, + { + "id": "perf-002", + "content": "For finding kth largest/smallest, consider using heapq.nlargest/nsmallest for better performance on large datasets", + "tags": ["performance", "algorithms", "advanced"], + "confidence": 0.75, + "added_epoch": 3, + "note": "Applicable for lists > 100 elements" + } + ] + } + }, + "metadata": { + "total_bullets": 18, + "sections_count": 5, + "avg_confidence": 0.87, + "tags_frequency": { + "edge-cases": 6, + "validation": 5, + "best-practices": 4, + "type-checking": 3, + "pythonic": 2, + "performance": 2, + "algorithms": 3, + "error-handling": 1, + "documentation": 1, + "readability": 1, + "safety": 1, + "immutability": 1, + "return-values": 1, + "advanced": 1 + } + } +} diff --git a/docs/openspec/changes/update-eval-schema-v2/design.md b/docs/openspec/changes/update-eval-schema-v2/design.md index 21570d59c..16fe944e0 100644 --- a/docs/openspec/changes/update-eval-schema-v2/design.md +++ b/docs/openspec/changes/update-eval-schema-v2/design.md @@ -2,13 +2,13 @@ ## Context -The current eval schema (V1) uses a simple structure with `testcases`, flat `messages` arrays, and implicit configuration inheritance. As AgentEvo evolves to support advanced features like ACE optimization, conversation threading, and template-based evaluation, we need a more structured schema that: +The current eval schema (V1) uses a simple structure with `testcases`, flat `messages` arrays, and implicit configuration inheritance. As AgentEvo evolves to support advanced features like conversation threading and template-based evaluation, we need a more structured schema that: - Explicitly declares execution configuration at appropriate levels - Supports conversation-based organization for multi-turn scenarios -- Enables pluggable optimization frameworks (starting with ACE) - Separates input messages from expected output messages - Aligns with modern eval framework conventions (similar to LangSmith, Braintrust, etc.) +- Provides clean separation between eval definitions and optimization configs **Stakeholders**: AgentEvo users, ACE framework integrators, eval pipeline maintainers @@ -21,18 +21,17 @@ The current eval schema (V1) uses a simple structure with `testcases`, flat `mes ## Goals / Non-Goals **Goals**: -- Define V2 schema supporting conversation threading, execution config, and optimization -- Enable ACE optimization framework integration +- Define V2 schema supporting conversation threading and execution config - Support template-based evaluators with configurable models +- Support multiple evaluators per eval case - Provide migration guide documenting V1 to V2 changes - Reject V1 format with clear error message and migration guidance **Non-Goals**: - Backward compatibility with V1 format (clean break) - Automatic conversion of V1 files to V2 (user responsibility) -- Full ACE implementation in this change (stub the integration points only) +- ACE or other optimization framework integration (separate config files, separate change) - Multi-language support in templates (English only for now) -- Custom optimization frameworks beyond ACE (future work) ## Decisions @@ -70,12 +69,19 @@ execution: - name: regex_validation # Third evaluator type: code # Code-based regex or keyword checking script: ./scripts/validate_format.py - optimization: - type: ace # Future: genetic, bayesian, etc. - playbook_path: ./playbooks/triage.json - max_epochs: 2 - max_reflector_rounds: 2 - allow_dynamic_sections: true +``` + +**Optimization Separation**: ACE and other optimization frameworks will use separate config files (e.g., `opts/ace-code-generation.yaml`) that reference eval files: +```yaml +# opts/ace-code-generation.yaml +type: ace +eval_files: + - evals/code-generation.test.yaml + - evals/code-review.test.yaml +playbook_path: ./playbooks/code-generation.json +max_epochs: 5 +max_reflector_rounds: 3 +allow_dynamic_sections: true ``` **Alternatives considered**: @@ -179,16 +185,6 @@ evalcases: - Version bump to indicate breaking change (e.g., 0.x → 1.0 or major version) - Consider providing conversion script to automate migration -### Risk: ACE Integration Complexity - -**Impact**: Medium - ACE framework may have different requirements than stubbed interface - -**Mitigation**: -- Design optimization config as extensible dict (ACE can add custom fields) -- Keep optimization block optional (defaults to no optimization) -- Defer full ACE implementation to separate change -- Coordinate with ACE team on config schema before implementation - ### Risk: Template Rendering Performance **Impact**: Low - Loading/rendering templates per eval case could slow down large suites @@ -246,6 +242,6 @@ If critical issues arise: - Proposal: Add `match_mode: prefix|exact|any` to evaluator config - Rationale: Defer to separate change on evaluator improvements -4. **Should optimization config support multiple strategies in one eval?** - - Proposal: No - one optimization type per eval case - - Rationale: Simplifies implementation, unclear use case for mixing strategies +4. **Should optimization configs (ACE, etc.) live in separate files or be part of eval schema?** + - Decision: Separate files in `opts/` directory that reference eval files + - Rationale: Clean separation of concerns - evals define test cases, optimization configs define how to improve prompts using those evals diff --git a/docs/openspec/changes/update-eval-schema-v2/proposal.md b/docs/openspec/changes/update-eval-schema-v2/proposal.md index 4e7555656..044bf26ee 100644 --- a/docs/openspec/changes/update-eval-schema-v2/proposal.md +++ b/docs/openspec/changes/update-eval-schema-v2/proposal.md @@ -2,13 +2,13 @@ ## Why -The current eval schema uses legacy naming (`testcases`, `messages`) and lacks support for advanced features like conversation threading, ACE optimization, and structured execution configuration. The new V2 schema aligns with modern eval frameworks, enabling: +The current eval schema uses legacy naming (`testcases`, `messages`) and lacks support for advanced features like conversation threading and structured execution configuration. The new V2 schema aligns with modern eval frameworks, enabling: - Conversation-based test organization via `conversation_id` -- Explicit execution configuration with target, evaluator, and optimization settings -- ACE (Automatic Cognitive Enhancement) integration for prompt optimization +- Explicit execution configuration with target and evaluator settings - Clearer separation between input and expected output messages - Better alignment with industry-standard eval formats +- Foundation for external optimization frameworks (ACE will reference evals, not be embedded in them) ## What Changes @@ -23,7 +23,8 @@ The current eval schema uses legacy naming (`testcases`, `messages`) and lacks s - `prompt`: Prompt template path (for llm_judge) - `model`: Model override (for llm_judge) - `script`: Script path (for code evaluators) - - `optimization`: ACE configuration with `type`, `playbook_path`, `max_epochs`, `max_reflector_rounds`, `allow_dynamic_sections` + +**Note**: ACE optimization configuration will be handled in separate optimization config files (e.g., `optimizers/*.yaml`) that reference eval files, not within eval cases themselves. ## Impact diff --git a/docs/openspec/changes/update-eval-schema-v2/tasks.md b/docs/openspec/changes/update-eval-schema-v2/tasks.md index 362825f9b..9e014a53c 100644 --- a/docs/openspec/changes/update-eval-schema-v2/tasks.md +++ b/docs/openspec/changes/update-eval-schema-v2/tasks.md @@ -14,7 +14,7 @@ - [ ] 2.2 Detect and reject V1 format (error if `testcases` found with migration guidance) - [ ] 2.3 Parse `evalcases` structure (error if neither `evalcases` nor `testcases` found) - [ ] 2.4 Parse `conversation_id` field (default to eval case id) -- [ ] 2.5 Parse `execution` block (target, evaluators array, optimization) +- [ ] 2.5 Parse `execution` block (target, evaluators array) - [ ] 2.5.1 Validate evaluator names are unique within a case - [ ] 2.5.2 Support different evaluator types (llm_judge, code) - [ ] 2.6 Parse `input_messages` array @@ -30,7 +30,6 @@ - [ ] 3.4.2 Collect scores from all evaluators with their unique names - [ ] 3.5 Add evaluator prompt loading and rendering (llm_judge) - [ ] 3.6 Add code evaluator execution (script-based, supports regex/keyword scripts) -- [ ] 3.7 Stub optimization framework integration points ## 4. Output Format Updates @@ -51,8 +50,7 @@ - [ ] 6.1 Update example file to V2 format (`docs/examples/simple/evals/example-v2.test.yaml`) - [ ] 6.2 Update README with V2 schema documentation - [ ] 6.3 Create migration guide documenting: `testcases` → `evalcases`, `messages` → `input_messages` + `expected_messages` -- [ ] 6.4 Document ACE optimization features -- [ ] 6.5 Update all internal eval files to V2 format +- [ ] 6.4 Update all internal eval files to V2 format ## 7. Testing From 3d01f3d98730f6fa69bcc3f77529f9d7ecf6a139 Mon Sep 17 00:00:00 2001 From: Christopher Tso Date: Sat, 15 Nov 2025 18:39:24 +1100 Subject: [PATCH 3/9] update specs --- docs/examples/simple/README.md | 2 +- docs/examples/simple/evals/example-v2.test.yaml | 3 ++- docs/openspec/changes/update-eval-schema-v2/design.md | 9 ++++++--- 3 files changed, 9 insertions(+), 5 deletions(-) diff --git a/docs/examples/simple/README.md b/docs/examples/simple/README.md index b6ebbea94..d90d01cf3 100644 --- a/docs/examples/simple/README.md +++ b/docs/examples/simple/README.md @@ -135,7 +135,7 @@ evalcases: # ... ``` -Groups related eval cases for analytics and optimization. +The `conversation_id` represents the full conversation that may be split into multiple eval cases. Most commonly, eval cases test the final response (e.g., `id: final-response`), but can also test intermediate conversation turns. This enables analytics and optimization at the conversation level. ### 4. File References diff --git a/docs/examples/simple/evals/example-v2.test.yaml b/docs/examples/simple/evals/example-v2.test.yaml index 3e745c4ff..dafaf3cd8 100644 --- a/docs/examples/simple/evals/example-v2.test.yaml +++ b/docs/examples/simple/evals/example-v2.test.yaml @@ -65,7 +65,8 @@ evalcases: # ========================================== - id: code-gen-python-comprehensive - # conversation_id groups related eval cases for analytics and optimization + # conversation_id represents the full conversation that may be split into multiple eval cases + # Most commonly, eval cases test the final response, but could also test intermediate turns conversation_id: python-code-generation outcome: | diff --git a/docs/openspec/changes/update-eval-schema-v2/design.md b/docs/openspec/changes/update-eval-schema-v2/design.md index 16fe944e0..224b4c264 100644 --- a/docs/openspec/changes/update-eval-schema-v2/design.md +++ b/docs/openspec/changes/update-eval-schema-v2/design.md @@ -132,18 +132,21 @@ expected_messages: # Expected output messages ### Decision 4: Conversation ID for Grouping -**Rationale**: Enables grouping related eval cases (e.g., variations of same scenario, multi-step workflows). Useful for: +**Rationale**: The `conversation_id` represents the full conversation that may be split into multiple eval cases. This enables: +- Testing different turns within the same conversation (e.g., intermediate responses vs final response) - Analytics (aggregate scores by conversation) -- Optimization (treat related cases as cohort) +- Optimization (treat conversation-level performance as a cohort) - Reporting (group results in UI) +Most commonly, eval cases test the final response (e.g., `id: final-response`), but conversations with multiple important decision points may have eval cases for intermediate turns. + **Usage**: ```yaml evalcases: - id: incident-triage-step-1 conversation_id: incident-triage-playbook ... - - id: incident-triage-step-2 + - id: incident-triage-playbook-final # Most common: test the final output conversation_id: incident-triage-playbook ... ``` From 1a0771390a09632fd60d4cd4943d269190e9b48b Mon Sep 17 00:00:00 2001 From: Christopher Tso Date: Sat, 15 Nov 2025 18:45:20 +1100 Subject: [PATCH 4/9] update spec --- .../changes/update-eval-schema-v2/design.md | 46 +++---------------- .../changes/update-eval-schema-v2/tasks.md | 7 ++- 2 files changed, 10 insertions(+), 43 deletions(-) diff --git a/docs/openspec/changes/update-eval-schema-v2/design.md b/docs/openspec/changes/update-eval-schema-v2/design.md index 224b4c264..5676db523 100644 --- a/docs/openspec/changes/update-eval-schema-v2/design.md +++ b/docs/openspec/changes/update-eval-schema-v2/design.md @@ -24,12 +24,10 @@ The current eval schema (V1) uses a simple structure with `testcases`, flat `mes - Define V2 schema supporting conversation threading and execution config - Support template-based evaluators with configurable models - Support multiple evaluators per eval case -- Provide migration guide documenting V1 to V2 changes -- Reject V1 format with clear error message and migration guidance **Non-Goals**: - Backward compatibility with V1 format (clean break) -- Automatic conversion of V1 files to V2 (user responsibility) +- Automatic conversion of V1 files to V2 (no users to migrate) - ACE or other optimization framework integration (separate config files, separate change) - Multi-language support in templates (English only for now) @@ -44,7 +42,7 @@ The current eval schema (V1) uses a simple structure with `testcases`, flat `mes - Use `scenarios`: Rejected - too generic, conflicts with requirement scenarios - Use `cases`: Rejected - ambiguous without context -**Migration**: V1 format (files with `testcases` key) will be rejected with a clear error message directing users to the migration guide. +**Migration**: V1 format (files with `testcases` key) will be rejected with a clear error message. ### Decision 2: Execution Block Structure @@ -163,8 +161,8 @@ evalcases: **Rationale**: - Simplifies implementation (no dual parser maintenance) - Clearer codebase without legacy support -- AgentEvo is early-stage; limited existing eval files to migrate -- Breaking change is acceptable at this maturity level +- No existing users to migrate +- Clean start with V2 schema only **Alternatives considered**: - Auto-detection with V1 support: Rejected - adds complexity for minimal benefit @@ -178,16 +176,6 @@ evalcases: ## Risks / Trade-offs -### Risk: Breaking Changes in Production Pipelines - -**Impact**: High - Users with automated eval pipelines will break if they upgrade without updating files - -**Mitigation**: -- Document as BREAKING CHANGE in release notes -- Provide clear migration guide with before/after examples -- Version bump to indicate breaking change (e.g., 0.x → 1.0 or major version) -- Consider providing conversion script to automate migration - ### Risk: Template Rendering Performance **Impact**: Low - Loading/rendering templates per eval case could slow down large suites @@ -203,33 +191,13 @@ evalcases: **Reasoning**: AgentEvo is evolving toward production-grade eval platform. Power users need fine-grained control, and clear structure aids maintainability. Beginners can use defaults and ignore advanced features. -## Migration Plan - -### Implementation (Current Change) +## Implementation Plan 1. Replace existing YAML parser with V2-only implementation 2. Update type definitions to V2 schema 3. Update all bundled examples to V2 format -4. Document V1→V2 migration guide with field mappings -5. Update CLI help text and error messages for V2 - -### User Migration Steps - -1. Review migration guide in documentation -2. Update eval files: - - `testcases` → `evalcases` - - `messages` → `input_messages` + `expected_messages` - - Add optional `conversation_id`, `execution` blocks as needed -3. Test updated files with `agentevo eval --dry-run` -4. Run production evals - -### Rollback Plan - -If critical issues arise: -- Revert to previous version of agentevo (with V1 support) -- Users can stay on old version until issues resolved -- Fix V2 parser and re-release -- Note: Once upgraded, users cannot downgrade without reverting their eval files to V1 format +4. Update CLI help text and error messages for V2 +5. Add validation to reject any V1 format files with clear error ## Open Questions diff --git a/docs/openspec/changes/update-eval-schema-v2/tasks.md b/docs/openspec/changes/update-eval-schema-v2/tasks.md index 9e014a53c..dba52da35 100644 --- a/docs/openspec/changes/update-eval-schema-v2/tasks.md +++ b/docs/openspec/changes/update-eval-schema-v2/tasks.md @@ -47,10 +47,9 @@ ## 6. Documentation & Examples -- [ ] 6.1 Update example file to V2 format (`docs/examples/simple/evals/example-v2.test.yaml`) -- [ ] 6.2 Update README with V2 schema documentation -- [ ] 6.3 Create migration guide documenting: `testcases` → `evalcases`, `messages` → `input_messages` + `expected_messages` -- [ ] 6.4 Update all internal eval files to V2 format +- [ ] 6.1 Update README with V2 schema documentation +- [ ] 6.2 Create migration guide documenting: `testcases` → `evalcases`, `messages` → `input_messages` + `expected_messages` +- [ ] 6.3 Update all internal eval files to V2 format ## 7. Testing From 10785fc49dd44fee95ddc74ba0f9fdfff38c7139 Mon Sep 17 00:00:00 2001 From: Christopher Tso Date: Sat, 15 Nov 2025 19:04:22 +1100 Subject: [PATCH 5/9] support v2 yaml format --- .../changes/update-eval-schema-v2/tasks.md | 24 ++++----- packages/core/src/evaluation/orchestrator.ts | 2 + packages/core/src/evaluation/types.ts | 2 + packages/core/src/evaluation/yaml-parser.ts | 51 ++++++++++++++----- 4 files changed, 54 insertions(+), 25 deletions(-) diff --git a/docs/openspec/changes/update-eval-schema-v2/tasks.md b/docs/openspec/changes/update-eval-schema-v2/tasks.md index dba52da35..c8d1f3490 100644 --- a/docs/openspec/changes/update-eval-schema-v2/tasks.md +++ b/docs/openspec/changes/update-eval-schema-v2/tasks.md @@ -2,7 +2,7 @@ ## 1. Type System Updates -- [ ] 1.1 Define new `EvalCase` type with `conversation_id`, `execution`, `input_messages`, `expected_messages` +- [x] 1.1 Define new `EvalCase` type with `conversation_id`, `execution`, `input_messages`, `expected_messages` (MVP: added conversation_id to TestCase) - [ ] 1.2 Define `ExecutionConfig` type with `target`, `evaluators`, `optimization` - [ ] 1.3 Define `EvaluatorConfig` type with `name`, `type`, `prompt`, `model`, `script` - [ ] 1.4 Define `OptimizationConfig` type with ACE parameters @@ -10,20 +10,20 @@ ## 2. Schema Parser Updates -- [ ] 2.1 Replace V1 parser with V2-only implementation +- [x] 2.1 Replace V1 parser with V2-only implementation (MVP: supports both V1 and V2 with fallback) - [ ] 2.2 Detect and reject V1 format (error if `testcases` found with migration guidance) -- [ ] 2.3 Parse `evalcases` structure (error if neither `evalcases` nor `testcases` found) -- [ ] 2.4 Parse `conversation_id` field (default to eval case id) +- [x] 2.3 Parse `evalcases` structure (error if neither `evalcases` nor `testcases` found) +- [x] 2.4 Parse `conversation_id` field (default to eval case id) - [ ] 2.5 Parse `execution` block (target, evaluators array) - [ ] 2.5.1 Validate evaluator names are unique within a case - [ ] 2.5.2 Support different evaluator types (llm_judge, code) -- [ ] 2.6 Parse `input_messages` array -- [ ] 2.7 Parse `expected_messages` array +- [x] 2.6 Parse `input_messages` array +- [x] 2.7 Parse `expected_messages` array ## 3. Execution Engine Updates -- [ ] 3.1 Update orchestrator to handle V2 schema -- [ ] 3.2 Support `conversation_id` for grouping eval cases +- [x] 3.1 Update orchestrator to handle V2 schema (conversation_id flows through) +- [x] 3.2 Support `conversation_id` for grouping eval cases - [ ] 3.3 Implement execution block resolution (case-level → file-level → default) - [ ] 3.4 Add support for multiple evaluators per case - [ ] 3.4.1 Execute evaluators in parallel when possible @@ -33,11 +33,11 @@ ## 4. Output Format Updates -- [ ] 4.1 Update `EvaluationResult` to include `conversation_id` +- [x] 4.1 Update `EvaluationResult` to include `conversation_id` - [ ] 4.2 Add `execution_config` to result metadata - [ ] 4.2.1 Support multiple scores from different evaluators (scores object with evaluator names as keys) -- [ ] 4.3 Update JSONL writer for new result structure -- [ ] 4.4 Update YAML writer for new result structure +- [x] 4.3 Update JSONL writer for new result structure (conversation_id included automatically) +- [x] 4.4 Update YAML writer for new result structure (conversation_id included automatically) ## 5. CLI Updates @@ -59,7 +59,7 @@ - [ ] 7.3.1 Add multiple evaluators execution tests - [ ] 7.3.2 Add test for different evaluator types (llm_judge, code) - [ ] 7.3.3 Add test for evaluator name uniqueness validation -- [ ] 7.4 Validate example-v2.test.yaml executes successfully +- [x] 7.4 Validate example-v2.test.yaml executes successfully (verified with manual test) - [ ] 7.5 Add error test for V1 format (should fail with clear message) ## 8. Validation & Deployment diff --git a/packages/core/src/evaluation/orchestrator.ts b/packages/core/src/evaluation/orchestrator.ts index 65a4e853b..d00f4a9e5 100644 --- a/packages/core/src/evaluation/orchestrator.ts +++ b/packages/core/src/evaluation/orchestrator.ts @@ -358,6 +358,7 @@ export async function runTestCase(options: RunTestCaseOptions): Promise isTestMessage(msg)); - if (messages.length === 0) { - logWarning(`Skipping test case with no valid messages: ${id}`); - continue; - } - - const userMessages = messages.filter((message) => message.role === "user"); - const assistantMessages = messages.filter((message) => message.role === "assistant"); + // In V2 format: input_messages contains system/user, expected_messages contains assistant + // In V1 format: messages contains all roles including assistant + const inputMessages = inputMessagesValue.filter((msg): msg is TestMessage => isTestMessage(msg)); + const expectedMessages = expectedMessagesValue.filter((msg): msg is TestMessage => isTestMessage(msg)); + + // For V1 compatibility: extract assistant messages from inputMessages if no expectedMessages + const assistantMessages = expectedMessages.length > 0 + ? expectedMessages.filter((message) => message.role === "assistant") + : inputMessages.filter((message) => message.role === "assistant"); + + const userMessages = inputMessages.filter((message) => message.role === "user"); if (assistantMessages.length === 0) { logWarning(`No assistant message found for test case: ${id}`); @@ -211,6 +235,7 @@ export async function loadTestCases( const testCase: TestCase = { id, + conversation_id: conversationId, task: userTextPrompt, user_segments: userSegments, expected_assistant_raw: expectedAssistantRaw, From 4506a1efc784259974c61825c9a99513f67645b5 Mon Sep 17 00:00:00 2001 From: Christopher Tso Date: Sat, 15 Nov 2025 19:45:16 +1100 Subject: [PATCH 6/9] add version to yaml schema --- .gitignore | 2 +- docs/examples/simple/.agentevo/targets.yaml | 63 +++++++++-------- .../simple/evals/example-v2.test.yaml | 5 +- .../changes/update-eval-schema-v2/tasks.md | 10 +-- .../src/evaluation/providers/targets-file.ts | 37 +++++++++- packages/core/src/evaluation/yaml-parser.ts | 68 +++++++++++-------- 6 files changed, 116 insertions(+), 69 deletions(-) diff --git a/.gitignore b/.gitignore index eb0231c70..59fadf0c9 100644 --- a/.gitignore +++ b/.gitignore @@ -8,5 +8,5 @@ coverage/ *.tsbuildinfo *.tsbuildinfo.* .DS_Store -.bbeval/ +.agentevo/ .env diff --git a/docs/examples/simple/.agentevo/targets.yaml b/docs/examples/simple/.agentevo/targets.yaml index dabb1c87d..f15959fee 100644 --- a/docs/examples/simple/.agentevo/targets.yaml +++ b/docs/examples/simple/.agentevo/targets.yaml @@ -1,37 +1,40 @@ +version: 2.0 + # A list of all supported evaluation targets for the project. # Each target defines a provider and its specific configuration. # Actual values for paths/keys are stored in the local .env file. -- name: vscode_projectx - provider: vscode - settings: - # VS Code provider requires a workspace template path. - workspace_template: PROJECTX_WORKSPACE_PATH - # This key makes the judge model configurable and is good practice - judge_target: azure_base +targets: + - name: vscode_projectx + provider: vscode + settings: + # VS Code provider requires a workspace template path. + workspace_template: PROJECTX_WORKSPACE_PATH + # This key makes the judge model configurable and is good practice + judge_target: azure_base -- name: vscode_insiders_projectx - provider: vscode-insiders - settings: - # VS Code Insiders provider (preview version) requires a workspace template path. - workspace_template: PROJECTX_WORKSPACE_PATH - # This key makes the judge model configurable and is good practice - judge_target: azure_base + - name: vscode_insiders_projectx + provider: vscode-insiders + settings: + # VS Code Insiders provider (preview version) requires a workspace template path. + workspace_template: PROJECTX_WORKSPACE_PATH + # This key makes the judge model configurable and is good practice + judge_target: azure_base -- name: azure_base - provider: azure - # The base LLM provider needs no extra settings, as the model is - # defined in the .env file. - settings: - endpoint: AZURE_OPENAI_ENDPOINT - api_key: AZURE_OPENAI_API_KEY - model: AZURE_DEPLOYMENT_NAME + - name: azure_base + provider: azure + # The base LLM provider needs no extra settings, as the model is + # defined in the .env file. + settings: + endpoint: AZURE_OPENAI_ENDPOINT + api_key: AZURE_OPENAI_API_KEY + model: AZURE_DEPLOYMENT_NAME -- name: default - provider: azure - # The base LLM provider needs no extra settings, as the model is - # defined in the .env file. - settings: - endpoint: AZURE_OPENAI_ENDPOINT - api_key: AZURE_OPENAI_API_KEY - model: AZURE_DEPLOYMENT_NAME + - name: default + provider: azure + # The base LLM provider needs no extra settings, as the model is + # defined in the .env file. + settings: + endpoint: AZURE_OPENAI_ENDPOINT + api_key: AZURE_OPENAI_API_KEY + model: AZURE_DEPLOYMENT_NAME \ No newline at end of file diff --git a/docs/examples/simple/evals/example-v2.test.yaml b/docs/examples/simple/evals/example-v2.test.yaml index dafaf3cd8..95bd09967 100644 --- a/docs/examples/simple/evals/example-v2.test.yaml +++ b/docs/examples/simple/evals/example-v2.test.yaml @@ -1,6 +1,7 @@ # AgentEvo V2 Eval Schema Example # Demonstrates all V2 schema features with real file references and minimal redundancy +version: 2.0 description: V2 schema example showing basic features, conversation threading, multiple evaluators, and ACE optimization # File-level default target (optional) @@ -42,7 +43,7 @@ evalcases: # File references are resolved relative to the eval file directory # Files ending in .instructions.md are treated as guidelines - type: file - value: ../evaluators/prompts/javascript.instructions.md + value: ../prompts/javascript.instructions.md # V2 Schema: Use 'expected_messages' instead of inline assistant message in V1 expected_messages: @@ -102,7 +103,7 @@ evalcases: 4. Raises appropriate exceptions for invalid input # Reference to real instruction file - type: file - value: ../evaluators/prompts/python.instructions.md + value: ../prompts/python.instructions.md expected_messages: - role: assistant diff --git a/docs/openspec/changes/update-eval-schema-v2/tasks.md b/docs/openspec/changes/update-eval-schema-v2/tasks.md index c8d1f3490..4fea744cf 100644 --- a/docs/openspec/changes/update-eval-schema-v2/tasks.md +++ b/docs/openspec/changes/update-eval-schema-v2/tasks.md @@ -10,8 +10,8 @@ ## 2. Schema Parser Updates -- [x] 2.1 Replace V1 parser with V2-only implementation (MVP: supports both V1 and V2 with fallback) -- [ ] 2.2 Detect and reject V1 format (error if `testcases` found with migration guidance) +- [x] 2.1 Replace V1 parser with V2-only implementation +- [x] 2.2 Detect and reject V1 format (error if `testcases` found with migration guidance) - [x] 2.3 Parse `evalcases` structure (error if neither `evalcases` nor `testcases` found) - [x] 2.4 Parse `conversation_id` field (default to eval case id) - [ ] 2.5 Parse `execution` block (target, evaluators array) @@ -43,7 +43,7 @@ - [ ] 5.1 Update help text with V2 schema examples - [ ] 5.2 Add validation for V2-specific features -- [ ] 5.3 Ensure clear error messages when V1 format detected +- [x] 5.3 Ensure clear error messages when V1 format detected ## 6. Documentation & Examples @@ -60,11 +60,11 @@ - [ ] 7.3.2 Add test for different evaluator types (llm_judge, code) - [ ] 7.3.3 Add test for evaluator name uniqueness validation - [x] 7.4 Validate example-v2.test.yaml executes successfully (verified with manual test) -- [ ] 7.5 Add error test for V1 format (should fail with clear message) +- [x] 7.5 Add error test for V1 format (should fail with clear message) ## 8. Validation & Deployment - [ ] 8.1 Run `openspec validate update-eval-schema-v2 --strict` - [ ] 8.2 Run full test suite -- [ ] 8.3 Verify clear error message when V1 format detected +- [x] 8.3 Verify clear error message when V1 format detected - [ ] 8.4 Create release notes documenting breaking changes diff --git a/packages/core/src/evaluation/providers/targets-file.ts b/packages/core/src/evaluation/providers/targets-file.ts index db948142d..3791e892d 100644 --- a/packages/core/src/evaluation/providers/targets-file.ts +++ b/packages/core/src/evaluation/providers/targets-file.ts @@ -9,6 +9,34 @@ function isRecord(value: unknown): value is Record { return typeof value === "object" && value !== null && !Array.isArray(value); } +function checkVersion(parsed: Record, absolutePath: string): void { + const version = typeof parsed.version === 'number' ? parsed.version : + typeof parsed.version === 'string' ? parseFloat(parsed.version) : + undefined; + + if (version === undefined) { + throw new Error( + `Missing version field in targets.yaml at ${absolutePath}.\n` + + `Please add 'version: 2.0' at the top of the file.` + ); + } + + if (version < 2.0) { + throw new Error( + `Outdated targets.yaml format (version ${version}) at ${absolutePath}.\n` + + `Please update to version 2.0 format with 'targets' array.` + ); + } +} + +function extractTargetsArray(parsed: Record, absolutePath: string): unknown[] { + const targets = parsed.targets; + if (!Array.isArray(targets)) { + throw new Error(`targets.yaml at ${absolutePath} must have a 'targets' array`); + } + return targets; +} + function assertTargetDefinition(value: unknown, index: number, filePath: string): TargetDefinition { if (!isRecord(value)) { throw new Error(`targets.yaml entry at index ${index} in ${filePath} must be an object`); @@ -53,11 +81,14 @@ export async function readTargetDefinitions(filePath: string): Promise assertTargetDefinition(entry, index, absolutePath)); + checkVersion(parsed, absolutePath); + + const targets = extractTargetsArray(parsed, absolutePath); + const definitions = targets.map((entry, index) => assertTargetDefinition(entry, index, absolutePath)); return definitions; } diff --git a/packages/core/src/evaluation/yaml-parser.ts b/packages/core/src/evaluation/yaml-parser.ts index d27c844a7..bfa4a4454 100644 --- a/packages/core/src/evaluation/yaml-parser.ts +++ b/packages/core/src/evaluation/yaml-parser.ts @@ -51,8 +51,8 @@ type LoadOptions = { }; type RawTestSuite = JsonObject & { + readonly version?: JsonValue; readonly grader?: JsonValue; - readonly testcases?: JsonValue; readonly evalcases?: JsonValue; readonly target?: JsonValue; }; @@ -61,7 +61,6 @@ type RawTestCase = JsonObject & { readonly id?: JsonValue; readonly conversation_id?: JsonValue; readonly outcome?: JsonValue; - readonly messages?: JsonValue; readonly input_messages?: JsonValue; readonly expected_messages?: JsonValue; readonly grader?: JsonValue; @@ -92,14 +91,34 @@ export async function loadTestCases( } const suite = parsed as RawTestSuite; - // Support both V2 (evalcases) and V1 (testcases) for now - const rawTestcases = Array.isArray(suite.evalcases) - ? suite.evalcases - : Array.isArray(suite.testcases) - ? suite.testcases - : undefined; - if (!rawTestcases) { - throw new Error(`Invalid test file format: ${testFilePath} - missing 'evalcases' or 'testcases' field`); + + // Check version field to determine schema format + const version = typeof suite.version === 'number' ? suite.version : + typeof suite.version === 'string' ? parseFloat(suite.version) : + undefined; + + // Require V2 format (version 2.0 or higher) + if (version === undefined) { + throw new Error( + `Missing version field in ${testFilePath}.\n` + + `Please add 'version: 2.0' at the top of the file.` + ); + } + + if (version < 2.0) { + throw new Error( + `V1 eval format detected in ${testFilePath} (version ${version}).\n` + + `The eval schema has been updated to V2. Please migrate your file:\n` + + ` 1. Update to 'version: 2.0'\n` + + ` 2. Rename 'testcases' to 'evalcases'\n` + + ` 3. Split 'messages' into 'input_messages' and 'expected_messages'` + ); + } + + // V2 format: version 2.0 or higher + const rawTestcases = suite.evalcases; + if (!Array.isArray(rawTestcases)) { + throw new Error(`Invalid test file format: ${testFilePath} - missing 'evalcases' field`); } const globalGrader = coerceGrader(suite.grader) ?? "llm_judge"; @@ -116,31 +135,24 @@ export async function loadTestCases( const conversationId = asString(testcase.conversation_id); const outcome = asString(testcase.outcome); - // Support both V2 (input_messages + expected_messages) and V1 (messages) - const inputMessagesValue = Array.isArray(testcase.input_messages) - ? testcase.input_messages - : Array.isArray(testcase.messages) - ? testcase.messages - : undefined; - const expectedMessagesValue = Array.isArray(testcase.expected_messages) - ? testcase.expected_messages - : []; - - if (!id || !outcome || !inputMessagesValue) { + const inputMessagesValue = testcase.input_messages; + const expectedMessagesValue = testcase.expected_messages; + + if (!id || !outcome || !Array.isArray(inputMessagesValue)) { logWarning(`Skipping incomplete test case: ${id ?? "unknown"}`); continue; } + + if (!Array.isArray(expectedMessagesValue)) { + logWarning(`Test case '${id}' missing expected_messages array`); + continue; + } - // In V2 format: input_messages contains system/user, expected_messages contains assistant - // In V1 format: messages contains all roles including assistant + // V2 format: input_messages contains system/user, expected_messages contains assistant const inputMessages = inputMessagesValue.filter((msg): msg is TestMessage => isTestMessage(msg)); const expectedMessages = expectedMessagesValue.filter((msg): msg is TestMessage => isTestMessage(msg)); - // For V1 compatibility: extract assistant messages from inputMessages if no expectedMessages - const assistantMessages = expectedMessages.length > 0 - ? expectedMessages.filter((message) => message.role === "assistant") - : inputMessages.filter((message) => message.role === "assistant"); - + const assistantMessages = expectedMessages.filter((message) => message.role === "assistant"); const userMessages = inputMessages.filter((message) => message.role === "user"); if (assistantMessages.length === 0) { From 12250cf5d0d928745cdb337467190bd8ad55cefd Mon Sep 17 00:00:00 2001 From: Christopher Tso Date: Sat, 15 Nov 2025 19:51:44 +1100 Subject: [PATCH 7/9] update example eval --- docs/examples/simple/README.md | 10 +- ...example-v2.test.yaml => example-eval.yaml} | 4 +- .../simple/evals/example-v1.test.yaml | 197 ------------------ .../optimizers/ace-code-generation.yaml | 2 +- .../specs/evaluation/spec.md | 6 +- .../changes/update-eval-schema-v2/tasks.md | 4 +- 6 files changed, 13 insertions(+), 210 deletions(-) rename docs/examples/simple/evals/{example-v2.test.yaml => example-eval.yaml} (99%) delete mode 100644 docs/examples/simple/evals/example-v1.test.yaml diff --git a/docs/examples/simple/README.md b/docs/examples/simple/README.md index d90d01cf3..49b228b4f 100644 --- a/docs/examples/simple/README.md +++ b/docs/examples/simple/README.md @@ -7,7 +7,7 @@ This directory demonstrates AgentEvo's V2 eval schema with complete, working exa ``` simple/ ├── evals/ # Evaluation test cases -│ ├── example-v2.test.yaml # Main V2 schema example +│ ├── example-eval.yaml # Main V2 schema example │ └── *.test.yaml # Additional eval files ├── evaluators/ # Evaluator components │ ├── prompts/ # LLM judge prompt templates @@ -29,7 +29,7 @@ simple/ ### Evaluation Files (`evals/`) -- **`example-v2.test.yaml`**: Complete V2 schema demonstration showing: +- **`example-eval.yaml`**: Complete V2 schema demonstration showing: - Basic features: `input_messages`, `expected_messages` - File references and content blocks - Conversation threading with `conversation_id` @@ -71,10 +71,10 @@ simple/ agentevo eval evals/ # Run specific eval file -agentevo eval evals/example-v2.test.yaml +agentevo eval evals/example-eval.yaml # Run with specific target -agentevo eval evals/example-v2.test.yaml --target azure_base +agentevo eval evals/example-eval.yaml --target azure_base ``` ### With Optimization (Future) @@ -255,7 +255,7 @@ Key changes when updating existing eval files: ## Next Steps -- Review `example-v2.test.yaml` to understand the schema +- Review `example-eval.yaml` to understand the schema - Create your own eval cases following the V2 format - Write custom evaluator scripts for domain-specific validation - Create LLM judge templates for semantic evaluation diff --git a/docs/examples/simple/evals/example-v2.test.yaml b/docs/examples/simple/evals/example-eval.yaml similarity index 99% rename from docs/examples/simple/evals/example-v2.test.yaml rename to docs/examples/simple/evals/example-eval.yaml index 95bd09967..85f4db770 100644 --- a/docs/examples/simple/evals/example-v2.test.yaml +++ b/docs/examples/simple/evals/example-eval.yaml @@ -6,7 +6,7 @@ description: V2 schema example showing basic features, conversation threading, m # File-level default target (optional) # Can be overridden per eval case via execution.target -target: azure_base +target: default # V2 Schema: Use 'evalcases' instead of 'testcases' (V1) evalcases: @@ -76,7 +76,7 @@ evalcases: # V2 Schema: Execution block for per-case configuration execution: # Override file-level target for this specific eval case - target: default_internal + target: azure_base # Multiple evaluators - supports both code-based and LLM judges evaluators: diff --git a/docs/examples/simple/evals/example-v1.test.yaml b/docs/examples/simple/evals/example-v1.test.yaml deleted file mode 100644 index ff27cb2ff..000000000 --- a/docs/examples/simple/evals/example-v1.test.yaml +++ /dev/null @@ -1,197 +0,0 @@ -description: Sample test cases demonstrating various eval patterns and content types -grader: llm_judge -target: azure_base - -testcases: -- id: simple-text-conversation - outcome: AI responds with helpful code analysis and mentions SUPERSECRET_INSTRUCTION_MARKER_JAVASCRIPT - messages: - - role: system - content: You are an expert software developer who provides clear, concise code reviews. - - role: user - content: - - type: text - - value: |- - Please review this JavaScript function: - - ```javascript - function calculateTotal(items) { - let total = 0; - for (let i = 0; i < items.length; i++) { - total += items[i].price * items[i].quantity; - } - return total; - } - ``` - - type: file - value: /docs/examples/simple/prompts/javascript.instructions.md - - role: assistant - content: |- - The function looks good overall. Here's my analysis: - - **Strengths:** - - Clear naming and purpose - - Handles array iteration correctly - - Simple and readable logic - - **Suggestions:** - - Consider using `reduce()` for a more functional approach - - Add input validation for edge cases (null/undefined items) - - Consider using `const` instead of `let` for the total variable if using reduce - -- id: multi-turn-debugging-conversation - outcome: AI provides step-by-step debugging assistance and mentions SUPERSECRET_INSTRUCTION_MARKER_PYTHON - messages: - - role: system - content: You are a debugging expert who helps identify and fix code issues systematically. Always follow the provided coding guidelines. - - role: user - content: - - type: text - value: |- - My Python script is throwing an error. Here's the code: - - ```python - def process_files(file_list): - results = [] - for file in file_list: - with open(file, 'r') as f: - data = f.read() - results.append(len(data)) - return results - - files = ['data.txt', 'config.json', 'missing.txt'] - print(process_files(files)) - ``` - - Error: `FileNotFoundError: [Errno 2] No such file or directory: 'missing.txt'` - - type: file - value: /docs/examples/simple/prompts/python.instructions.md - - role: assistant - content: |- - The error occurs because `missing.txt` doesn't exist. Following the coding guidelines (SUPERSECRET_INSTRUCTION_MARKER_PYTHON), here's how to fix it with proper error handling: - - **Option 1: Skip missing files** - ```python - import os - - def process_files(file_list): - results = [] - for file in file_list: - if os.path.exists(file): - with open(file, 'r') as f: - data = f.read() - results.append(len(data)) - else: - print(f"Warning: {file} not found, skipping...") - return results - ``` - - **Option 2: Handle with try/except** - ```python - def process_files(file_list): - results = [] - for file in file_list: - try: - with open(file, 'r') as f: - data = f.read() - results.append(len(data)) - except FileNotFoundError: - print(f"Error: {file} not found") - results.append(0) # or handle as needed - return results - ``` - - Which approach would you prefer? - - role: user - content: I like option 2, but I want to return a dictionary with filename and size instead of just sizes. - - role: assistant - content: |- - Perfect! Here's the modified version that returns a dictionary, following proper error handling practices (SUPERSECRET_INSTRUCTION_MARKER_PYTHON): - - ```python - def process_files(file_list): - results = {} - for file in file_list: - try: - with open(file, 'r') as f: - data = f.read() - results[file] = len(data) - except FileNotFoundError: - print(f"Error: {file} not found") - results[file] = None # or 0, or skip entirely - except Exception as e: - print(f"Error reading {file}: {e}") - results[file] = None - return results - - files = ['data.txt', 'config.json', 'missing.txt'] - result = process_files(files) - print(result) - # Output might be: {'data.txt': 150, 'config.json': 75, 'missing.txt': None} - ``` - - This also handles other potential file reading errors beyond just missing files. - -- id: code-generation-with-constraints - outcome: AI generates code following specific requirements and constraints - note: Tests ability to follow detailed specifications - messages: - - role: system - content: |- - You are a code generator that follows specifications exactly. Always include: - - Input validation - - Error handling - - Clear documentation - - Type hints (for Python) - - role: user - content: |- - Create a Python function that: - 1. Takes a list of integers - 2. Returns the second largest number - 3. Handles edge cases (empty list, single item, duplicates) - 4. Raises appropriate exceptions for invalid input - - role: assistant - content: |- - ```python - from typing import List, Union - - def find_second_largest(numbers: List[int]) -> Union[int, None]: - """ - Find the second largest number in a list of integers. - - Args: - numbers: List of integers to search - - Returns: - The second largest unique number, or None if not available - - Raises: - TypeError: If input is not a list or contains non-integers - ValueError: If list is empty - """ - # Input validation - if not isinstance(numbers, list): - raise TypeError("Input must be a list") - - if not numbers: - raise ValueError("List cannot be empty") - - if not all(isinstance(x, int) for x in numbers): - raise TypeError("All elements must be integers") - - # Handle edge cases - unique_numbers = list(set(numbers)) - - if len(unique_numbers) < 2: - return None # Not enough unique numbers - - # Sort and return second largest - unique_numbers.sort(reverse=True) - return unique_numbers[1] - - - # Example usage: - # find_second_largest([1, 3, 2, 3, 4]) # Returns 3 - # find_second_largest([5, 5, 5]) # Returns None - # find_second_largest([]) # Raises ValueError - ``` \ No newline at end of file diff --git a/docs/examples/simple/optimizers/ace-code-generation.yaml b/docs/examples/simple/optimizers/ace-code-generation.yaml index 52885c358..18e58f9cf 100644 --- a/docs/examples/simple/optimizers/ace-code-generation.yaml +++ b/docs/examples/simple/optimizers/ace-code-generation.yaml @@ -9,7 +9,7 @@ type: ace # Eval files to use for optimization # ACE will run these evals to measure prompt performance and guide improvements eval_files: - - ../evals/example-v2.test.yaml + - ../evals/example-eval.yaml # - ../evals/code-generation-edge-cases.test.yaml # - ../evals/code-review-security.test.yaml diff --git a/docs/openspec/changes/update-eval-schema-v2/specs/evaluation/spec.md b/docs/openspec/changes/update-eval-schema-v2/specs/evaluation/spec.md index 229a10f47..9afcbf6ef 100644 --- a/docs/openspec/changes/update-eval-schema-v2/specs/evaluation/spec.md +++ b/docs/openspec/changes/update-eval-schema-v2/specs/evaluation/spec.md @@ -2,7 +2,7 @@ ### Requirement: V2 Schema Format -The system SHALL support the V2 eval schema format as documented in `docs/examples/simple/evals/example-v2.test.yaml`. +The system SHALL support the V2 eval schema format as documented in `docs/examples/simple/evals/example-eval.yaml`. #### Scenario: Parse V2 schema with all features @@ -19,7 +19,7 @@ The system SHALL support the V2 eval schema format as documented in `docs/exampl #### Scenario: V2 example file executes successfully -- **WHEN** the system runs `docs/examples/simple/evals/example-v2.test.yaml` +- **WHEN** the system runs `docs/examples/simple/evals/example-eval.yaml` - **THEN** all four example eval cases execute without errors - **AND** demonstrates simple eval, multi-turn conversation, ACE optimization, and code-based evaluation - **AND** produces valid results in the specified output format @@ -187,7 +187,7 @@ The system SHALL successfully execute the bundled V2 example evaluation file to #### Scenario: Execute V2 example demonstrating all features -- **WHEN** the system runs `docs/examples/simple/evals/example-v2.test.yaml` (V2 format) +- **WHEN** the system runs `docs/examples/simple/evals/example-eval.yaml` (V2 format) - **THEN** all eval cases execute successfully as documented in the example file - **AND** demonstrates all V2 features with inline YAML comments explaining each capability diff --git a/docs/openspec/changes/update-eval-schema-v2/tasks.md b/docs/openspec/changes/update-eval-schema-v2/tasks.md index 4fea744cf..4331f5153 100644 --- a/docs/openspec/changes/update-eval-schema-v2/tasks.md +++ b/docs/openspec/changes/update-eval-schema-v2/tasks.md @@ -6,7 +6,7 @@ - [ ] 1.2 Define `ExecutionConfig` type with `target`, `evaluators`, `optimization` - [ ] 1.3 Define `EvaluatorConfig` type with `name`, `type`, `prompt`, `model`, `script` - [ ] 1.4 Define `OptimizationConfig` type with ACE parameters -- [ ] 1.5 Remove legacy `TestCase` type and V1 schema support +- [x] 1.5 Remove legacy `TestCase` type and V1 schema support ## 2. Schema Parser Updates @@ -59,7 +59,7 @@ - [ ] 7.3.1 Add multiple evaluators execution tests - [ ] 7.3.2 Add test for different evaluator types (llm_judge, code) - [ ] 7.3.3 Add test for evaluator name uniqueness validation -- [x] 7.4 Validate example-v2.test.yaml executes successfully (verified with manual test) +- [x] 7.4 Validate example-eval.yaml executes successfully (verified with manual test) - [x] 7.5 Add error test for V1 format (should fail with clear message) ## 8. Validation & Deployment From be668733fb1300bf4c2e7df861882cfe8552e191 Mon Sep 17 00:00:00 2001 From: Christopher Tso Date: Sat, 15 Nov 2025 19:52:11 +1100 Subject: [PATCH 8/9] update README --- docs/examples/simple/README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/examples/simple/README.md b/docs/examples/simple/README.md index 49b228b4f..96007e2a2 100644 --- a/docs/examples/simple/README.md +++ b/docs/examples/simple/README.md @@ -8,7 +8,7 @@ This directory demonstrates AgentEvo's V2 eval schema with complete, working exa simple/ ├── evals/ # Evaluation test cases │ ├── example-eval.yaml # Main V2 schema example -│ └── *.test.yaml # Additional eval files +│ └── *.yaml # Additional eval files ├── evaluators/ # Evaluator components │ ├── prompts/ # LLM judge prompt templates │ │ ├── code-correctness-judge.md # Semantic code evaluation From 40a6fb0d5c56a64b97b6732a166bb6d9f28f6491 Mon Sep 17 00:00:00 2001 From: Christopher Tso Date: Sat, 15 Nov 2025 19:55:27 +1100 Subject: [PATCH 9/9] bump version to 0.2.0 --- apps/cli/package.json | 2 +- packages/core/package.json | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/apps/cli/package.json b/apps/cli/package.json index 8a2eff5ca..bbcf536bb 100644 --- a/apps/cli/package.json +++ b/apps/cli/package.json @@ -1,6 +1,6 @@ { "name": "agentevo", - "version": "0.1.6", + "version": "0.2.0", "description": "CLI entry point for AgentEvo", "type": "module", "repository": { diff --git a/packages/core/package.json b/packages/core/package.json index 16bd71f3c..8fc1a4980 100644 --- a/packages/core/package.json +++ b/packages/core/package.json @@ -1,6 +1,6 @@ { "name": "@agentevo/core", - "version": "0.1.6", + "version": "0.2.0", "description": "Primitive runtime components for AgentEvo", "type": "module", "repository": {