Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
26 changes: 13 additions & 13 deletions docs/ai-python/content/docs/basics/model-operations.mdx
Original file line number Diff line number Diff line change
Expand Up @@ -255,37 +255,37 @@ are ordered from the highest score to the lowest.

## Evaluate state

Use `experimental_evaluate` to ask choice, score, and boolean questions about one shared state. Define matching Pydantic models for the questions and answers:
Use `ai.ops.experimental.evaluate` to ask choice, score, and boolean questions about one shared state. Define matching Pydantic models for the questions and answers:

```python
import pydantic


class Questions(pydantic.BaseModel):
department: ai.ops.ChoiceQuestion
severity: ai.ops.ScoreQuestion
requests_refund: ai.ops.BooleanQuestion
department: ai.ops.experimental.ChoiceQuestion
severity: ai.ops.experimental.ScoreQuestion
requests_refund: ai.ops.experimental.BooleanQuestion


class Answers(pydantic.BaseModel):
department: ai.ops.ChoiceAnswer
severity: ai.ops.ScoreAnswer
requests_refund: ai.ops.BooleanAnswer
department: ai.ops.experimental.ChoiceAnswer
severity: ai.ops.experimental.ScoreAnswer
requests_refund: ai.ops.experimental.BooleanAnswer


result = await ai.ops.experimental_evaluate(
result = await ai.ops.experimental.evaluate(
ai.get_model("typesafe-ai/jev"),
{"message": "Please refund the duplicate charge."},
Questions(
department=ai.ops.ChoiceQuestion(
department=ai.ops.experimental.ChoiceQuestion(
instructions="Which team should handle this?",
criteria={"billing": "Charges and refunds", "support": "Other"},
),
severity=ai.ops.ScoreQuestion(
severity=ai.ops.experimental.ScoreQuestion(
instructions="How severe is this?",
criteria=["Cosmetic", "Has workaround", "Blocking"],
),
requests_refund=ai.ops.BooleanQuestion(
requests_refund=ai.ops.experimental.BooleanQuestion(
instructions="Is the customer requesting a refund?",
),
),
Expand All @@ -299,11 +299,11 @@ When question IDs are dynamic, pass a mapping and omit `output_type`. The item
value is then a validated answer mapping:

```python
result = await ai.ops.experimental_evaluate(
result = await ai.ops.experimental.evaluate(
ai.get_model("typesafe-ai/jev"),
state,
{
question_id: ai.ops.BooleanQuestion(instructions=instructions)
question_id: ai.ops.experimental.BooleanQuestion(instructions=instructions)
for question_id, instructions in dynamic_questions.items()
},
)
Expand Down
11 changes: 7 additions & 4 deletions docs/ai-python/content/docs/reference/ops.mdx
Original file line number Diff line number Diff line change
Expand Up @@ -239,10 +239,13 @@ provider.
- `index`: Position of the document in the original input list.
- `score`: Relevance score for the query.

## experimental_evaluate
## experimental.evaluate

Evaluation questions, answers, `EvaluationInput`, and `EvaluationParams` live
under `ai.ops.experimental`.

```python
await ai.ops.experimental_evaluate(
await ai.ops.experimental.evaluate(
model,
state,
questions,
Expand All @@ -268,11 +271,11 @@ Pydantic questions return `Item[Answers]`, whose value is an instance of
`Item[dict[str, ChoiceAnswer | ScoreAnswer | BooleanAnswer]]`:

```python
result = await ai.ops.experimental_evaluate(
result = await ai.ops.experimental.evaluate(
model,
state,
{
question_id: ai.ops.BooleanQuestion(instructions=instructions)
question_id: ai.ops.experimental.BooleanQuestion(instructions=instructions)
for question_id, instructions in dynamic_questions.items()
},
)
Expand Down
32 changes: 16 additions & 16 deletions examples/models/gateway/evaluation.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,23 +17,23 @@


class RefundQuestions(pydantic.BaseModel):
requests_refund: ai.ops.BooleanQuestion
requests_refund: ai.ops.experimental.BooleanQuestion


class RefundAnswers(pydantic.BaseModel):
requests_refund: ai.ops.BooleanAnswer
requests_refund: ai.ops.experimental.BooleanAnswer


class TicketQuestions(pydantic.BaseModel):
queue: ai.ops.ChoiceQuestion
urgency: ai.ops.ScoreQuestion
refund_warranted: ai.ops.BooleanQuestion
queue: ai.ops.experimental.ChoiceQuestion
urgency: ai.ops.experimental.ScoreQuestion
refund_warranted: ai.ops.experimental.BooleanQuestion


class TicketAnswers(pydantic.BaseModel):
queue: ai.ops.ChoiceAnswer
urgency: ai.ops.ScoreAnswer
refund_warranted: ai.ops.BooleanAnswer
queue: ai.ops.experimental.ChoiceAnswer
urgency: ai.ops.experimental.ScoreAnswer
refund_warranted: ai.ops.experimental.BooleanAnswer


async def main() -> None:
Expand All @@ -44,11 +44,11 @@ async def main() -> None:

# Ask a single boolean question about plain text. Boolean answers are
# probabilities rather than only true or false.
refund_result = await ai.ops.experimental_evaluate(
refund_result = await ai.ops.experimental.evaluate(
model,
"Please refund the duplicate charge on my account.",
RefundQuestions(
requests_refund=ai.ops.BooleanQuestion(
requests_refund=ai.ops.experimental.BooleanQuestion(
instructions="Is the customer asking for a refund?",
)
),
Expand All @@ -62,7 +62,7 @@ async def main() -> None:

# Ask several question types about the same structured state. This is useful
# when related decisions should use exactly the same source information.
ticket: ai.ops.EvaluationInput = {
ticket: ai.ops.experimental.EvaluationInput = {
"message": (
"I was charged $240 twice for the same renewal. The service works, "
"but please refund the duplicate charge."
Expand All @@ -88,11 +88,11 @@ async def main() -> None:
"service_status": "operational",
}

result = await ai.ops.experimental_evaluate(
result = await ai.ops.experimental.evaluate(
model,
ticket,
TicketQuestions(
queue=ai.ops.ChoiceQuestion(
queue=ai.ops.experimental.ChoiceQuestion(
instructions="Which support queue should handle this ticket?",
criteria={
"billing": "Charges, duplicate payments, and refunds",
Expand All @@ -101,7 +101,7 @@ async def main() -> None:
"other": None,
},
),
urgency=ai.ops.ScoreQuestion(
urgency=ai.ops.experimental.ScoreQuestion(
instructions="How urgent is the customer's primary problem?",
criteria=[
"Low: no active customer impact",
Expand All @@ -110,7 +110,7 @@ async def main() -> None:
"Critical: security incident, outage, or ongoing loss",
],
),
refund_warranted=ai.ops.BooleanQuestion(
refund_warranted=ai.ops.experimental.BooleanQuestion(
instructions="Does the evidence warrant a refund?",
criteria={
"true": "A duplicate settlement or billing error is shown",
Expand All @@ -120,7 +120,7 @@ async def main() -> None:
),
output_type=TicketAnswers,
# Provider options are optional and apply only to this request.
params=ai.ops.EvaluationParams(
params=ai.ops.experimental.EvaluationParams(
provider_options={
"gateway": {
"zeroDataRetention": True,
Expand Down
24 changes: 2 additions & 22 deletions src/ai/ops/__init__.py
Original file line number Diff line number Diff line change
@@ -1,19 +1,8 @@
"""Model operations beyond LLM chat: media generation and friends."""

from . import experimental
from .audio import AudioParams, AudioPrompt, generate_audio
from .embeddings import EmbedParams, embed
from .evaluation import (
BooleanAnswer,
BooleanCriteria,
BooleanQuestion,
ChoiceAnswer,
ChoiceQuestion,
EvaluationInput,
EvaluationParams,
ScoreAnswer,
ScoreQuestion,
experimental_evaluate,
)
from .images import ImageParams, ImagePrompt, generate_image
from .items import Item, Warning
from .reranking import RankedDocument, RerankParams, rerank
Expand All @@ -28,30 +17,21 @@
__all__ = [
"AudioParams",
"AudioPrompt",
"BooleanAnswer",
"BooleanCriteria",
"BooleanQuestion",
"ChoiceAnswer",
"ChoiceQuestion",
"EmbedParams",
"EvaluationInput",
"EvaluationParams",
"FrameImage",
"ImageParams",
"ImagePrompt",
"Item",
"RankedDocument",
"RerankParams",
"ScoreAnswer",
"ScoreQuestion",
"TranscribeParams",
"Transcription",
"TranscriptionSegment",
"VideoParams",
"VideoPrompt",
"Warning",
"embed",
"experimental_evaluate",
"experimental",
"generate_audio",
"generate_image",
"generate_video",
Expand Down
27 changes: 27 additions & 0 deletions src/ai/ops/experimental/__init__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,27 @@
"""Experimental model operations."""

from .evaluation import (
BooleanAnswer,
BooleanCriteria,
BooleanQuestion,
ChoiceAnswer,
ChoiceQuestion,
EvaluationInput,
EvaluationParams,
ScoreAnswer,
ScoreQuestion,
evaluate,
)

__all__ = [
"BooleanAnswer",
"BooleanCriteria",
"BooleanQuestion",
"ChoiceAnswer",
"ChoiceQuestion",
"EvaluationInput",
"EvaluationParams",
"ScoreAnswer",
"ScoreQuestion",
"evaluate",
]
20 changes: 10 additions & 10 deletions src/ai/ops/evaluation.py → src/ai/ops/experimental/evaluation.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,16 +6,16 @@
import pydantic

class Questions(pydantic.BaseModel):
requests_refund: ai.ops.BooleanQuestion
requests_refund: ai.ops.experimental.BooleanQuestion

class Answers(pydantic.BaseModel):
requests_refund: ai.ops.BooleanAnswer
requests_refund: ai.ops.experimental.BooleanAnswer

result = await ai.ops.experimental_evaluate(
result = await ai.ops.experimental.evaluate(
ai.get_model("typesafe-ai/jev"),
{"message": "Please refund the duplicate charge."},
Questions(
requests_refund=ai.ops.BooleanQuestion(
requests_refund=ai.ops.experimental.BooleanQuestion(
instructions="Is the customer requesting a refund?",
),
),
Expand All @@ -32,11 +32,11 @@ class Answers(pydantic.BaseModel):

import pydantic

from .. import experimental_telemetry as telemetry
from . import items
from ... import experimental_telemetry as telemetry
from .. import items

if TYPE_CHECKING:
from ..models.core import model as model_
from ...models.core import model as model_


type EvaluationInput = (
Expand Down Expand Up @@ -160,7 +160,7 @@ class EvaluationParams:


@overload
async def experimental_evaluate[AnswerT: pydantic.BaseModel](
async def evaluate[AnswerT: pydantic.BaseModel](
model: model_.Model,
state: EvaluationInput,
questions: pydantic.BaseModel,
Expand All @@ -171,7 +171,7 @@ async def experimental_evaluate[AnswerT: pydantic.BaseModel](


@overload
async def experimental_evaluate(
async def evaluate(
model: model_.Model,
state: EvaluationInput,
questions: Mapping[str, pydantic.BaseModel],
Expand All @@ -181,7 +181,7 @@ async def experimental_evaluate(
) -> items.Item[dict[str, _Answer]]: ...


async def experimental_evaluate(
async def evaluate(
model: model_.Model,
state: EvaluationInput,
questions: pydantic.BaseModel | Mapping[str, pydantic.BaseModel],
Expand Down
22 changes: 11 additions & 11 deletions src/ai/providers/ai_gateway/protocol/v4.py
Original file line number Diff line number Diff line change
Expand Up @@ -535,22 +535,22 @@ async def stream(
async def evaluate(
gateway: gateway_client.GatewayClient,
model: models.Model,
state: ops.evaluation.EvaluationInput,
state: ops.experimental.EvaluationInput,
questions: Mapping[
str,
ops.evaluation.ChoiceQuestion
| ops.evaluation.ScoreQuestion
| ops.evaluation.BooleanQuestion,
ops.experimental.ChoiceQuestion
| ops.experimental.ScoreQuestion
| ops.experimental.BooleanQuestion,
],
*,
params: ops.evaluation.EvaluationParams,
params: ops.experimental.EvaluationParams,
) -> ops.items.Item[dict[str, Any]]:
"""Hit ``/evaluation-model`` and return raw evaluation answers."""
wire_questions: dict[str, dict[str, Any]] = {}
for question_id, question in questions.items():
wire_question = question.model_dump(mode="json", by_alias=True)
if (
isinstance(question, ops.evaluation.BooleanQuestion)
isinstance(question, ops.experimental.BooleanQuestion)
and question.criteria is None
):
wire_question.pop("criteria")
Expand Down Expand Up @@ -791,15 +791,15 @@ async def evaluate(
self,
client: gateway_client.GatewayClient,
model: models.Model,
state: ops.evaluation.EvaluationInput,
state: ops.experimental.EvaluationInput,
questions: Mapping[
str,
ops.evaluation.ChoiceQuestion
| ops.evaluation.ScoreQuestion
| ops.evaluation.BooleanQuestion,
ops.experimental.ChoiceQuestion
| ops.experimental.ScoreQuestion
| ops.experimental.BooleanQuestion,
],
*,
params: ops.evaluation.EvaluationParams,
params: ops.experimental.EvaluationParams,
provider: str,
) -> ops.items.Item[dict[str, Any]]:
_ = provider
Expand Down
2 changes: 1 addition & 1 deletion src/ai/providers/base.py
Original file line number Diff line number Diff line change
Expand Up @@ -27,13 +27,13 @@
from ..ops import (
audio,
embeddings,
evaluation,
images,
items,
reranking,
transcriptions,
videos,
)
from ..ops.experimental import evaluation
from ..types import events
from ..types import messages as messages_
from ..types import tools as tools_
Expand Down
Empty file.
Loading
Loading