From cff9d1f22270940dacd0f9c4f3c9de2ccdd5d694 Mon Sep 17 00:00:00 2001 From: Bernat Torres Date: Mon, 21 Sep 2026 15:07:53 +0200 Subject: [PATCH 01/30] feat(aio): add Jev boolean evaluations with TypeSafe keys --- .../internal/ai-observability-judge-inputs.md | 20 +++ .../costs/providers/manual-providers.test.ts | 1 + .../ai/costs/providers/manual-providers.ts | 4 + posthog/settings/web.py | 2 + .../eval_reports/report_agent/prompts.py | 9 +- .../ai_observability/evaluation_llm_judge.py | 59 ++++++++- .../ai_observability/evaluation_types.py | 1 + .../evaluation_workflow_activities.py | 2 + .../ai_observability/test_run_evaluation.py | 84 +++++++++++- .../backend/api/evaluation_config.py | 8 +- .../backend/api/evaluations.py | 6 + .../backend/api/provider_keys.py | 4 + .../ai_observability/backend/api/proxy.py | 5 +- .../ai_observability/backend/api/taggers.py | 6 +- .../api/test/test_evaluation_config.py | 14 ++ .../backend/api/test/test_evaluations.py | 27 +++- .../backend/api/test/test_provider_keys.py | 48 ++++++- .../ai_observability/backend/llm/client.py | 7 + .../backend/llm/test/test_typesafe.py | 122 ++++++++++++++++++ .../ai_observability/backend/llm/typesafe.py | 120 +++++++++++++++++ .../backend/models/provider_keys.py | 5 + .../ByokModelPickerNotice.stories.tsx | 47 +++++-- .../frontend/ByokModelPickerNotice.tsx | 5 +- .../ConversationDisplay/EvaluationDisplay.tsx | 14 +- .../EvaluationExplanation.stories.tsx | 21 +++ .../components/EvaluationExplanation.tsx | 24 ++++ .../components/GenerationEvalRunsTable.tsx | 13 +- .../evaluations/AIObservabilityEvaluation.tsx | 21 ++- .../components/EvaluationRunsTable.tsx | 13 +- .../frontend/evaluations/types.ts | 1 + .../frontend/generated/api.schemas.ts | 33 ++++- .../frontend/generated/api.zod.ts | 21 ++- .../frontend/modelPickerLogic.test.ts | 34 +++++ .../frontend/modelPickerLogic.ts | 92 +++++++++++-- .../playground/llmPlaygroundModelLogic.ts | 5 +- .../settings/LLMProviderKeysSettings.tsx | 6 +- .../frontend/settings/llmProviderKeysLogic.ts | 14 +- .../ai_observability/frontend/utils.test.ts | 9 ++ products/ai_observability/frontend/utils.ts | 7 +- services/mcp/src/api/generated.ts | 33 ++++- .../mcp/src/generated/ai_observability/api.ts | 7 +- 41 files changed, 863 insertions(+), 111 deletions(-) create mode 100644 products/ai_observability/backend/llm/test/test_typesafe.py create mode 100644 products/ai_observability/backend/llm/typesafe.py create mode 100644 products/ai_observability/frontend/components/EvaluationExplanation.stories.tsx create mode 100644 products/ai_observability/frontend/components/EvaluationExplanation.tsx diff --git a/docs/internal/ai-observability-judge-inputs.md b/docs/internal/ai-observability-judge-inputs.md index f2adaa6ca809..a8682d2eac3e 100644 --- a/docs/internal/ai-observability-judge-inputs.md +++ b/docs/internal/ai-observability-judge-inputs.md @@ -33,3 +33,23 @@ Generation evaluations extract message text without the per-message character cu They sample the combined input, tool definitions, and output only when that text exceeds 150,000 characters, with a final character slice enforcing the limit. Implementation: [trace judge](../../posthog/temporal/ai_observability/run_trace_evaluation.py), [session judge](../../posthog/temporal/ai_observability/run_session_evaluation.py), and [generation judge](../../posthog/temporal/ai_observability/evaluation_llm_judge.py). + +## Jev boolean judge + +Jev is available under the existing LLM judge option with a customer-provided TypeSafe API key. +Select the key and `jev-1.13.0` on each evaluation; TypeSafe keys cannot become the shared active provider key used by other AI features. +Provider keys keep the provider they were created with; switching providers requires a new key. +Jev supports boolean evaluations only and uses the same formatted text for generation, trace, and session targets. + +The evaluation prompt becomes a [Noul question](https://docs.typesafe.ai/primitives/noul). +A probability of at least 0.5 produces `true`; the evaluation's existing pass/fail polarity still applies. +For evaluations that allow N/A, a separate question checks whether the criteria apply, using the same threshold. +Uncertainty alone does not produce N/A. +The raw probability is stored in `$ai_evaluation_probability`, with token usage and the resolved model version. +Jev provides no written reasoning, so reports inspect the original source when explaining outcomes. + +TypeSafe rate limits and overload responses are retried through Temporal, honoring `Retry-After` up to five minutes. +If retries fail, the run fails and the evaluation stays enabled. +Invalid probabilities or missing answers fail the evaluation rather than producing a false result. +Inputs rejected for exceeding the model's context window are skipped. +See TypeSafe's [API reference](https://docs.typesafe.ai/api) and [model limits and pricing](https://docs.typesafe.ai/models). diff --git a/nodejs/src/ingestion/pipelines/ai/costs/providers/manual-providers.test.ts b/nodejs/src/ingestion/pipelines/ai/costs/providers/manual-providers.test.ts index f2824fe114c2..bdbdea4850c6 100644 --- a/nodejs/src/ingestion/pipelines/ai/costs/providers/manual-providers.test.ts +++ b/nodejs/src/ingestion/pipelines/ai/costs/providers/manual-providers.test.ts @@ -5,6 +5,7 @@ describe('manualCosts', () => { model: string expected: { prompt_token: number; completion_token: number; cache_read_token?: number } }> = [ + { model: 'jev-1.13.0', expected: { prompt_token: 0.000000042, completion_token: 0 } }, { model: 'gpt-4.5', expected: { diff --git a/nodejs/src/ingestion/pipelines/ai/costs/providers/manual-providers.ts b/nodejs/src/ingestion/pipelines/ai/costs/providers/manual-providers.ts index 0781a502ea6c..4085be25813d 100644 --- a/nodejs/src/ingestion/pipelines/ai/costs/providers/manual-providers.ts +++ b/nodejs/src/ingestion/pipelines/ai/costs/providers/manual-providers.ts @@ -1,6 +1,10 @@ import type { ModelCostRow } from './types' const manualProviderCosts: ModelCostRow[] = [ + { + model: 'jev-1.13.0', + cost: { default: { prompt_token: 0.000000042, completion_token: 0 } }, + }, { model: 'gpt-4.5', cost: { diff --git a/posthog/settings/web.py b/posthog/settings/web.py index b57420c28dac..3d797dc6c897 100644 --- a/posthog/settings/web.py +++ b/posthog/settings/web.py @@ -614,6 +614,8 @@ def static_varies_origin(headers, path, url): "TaskArtifactStatusEnum": ["active", "failed"], "RunSourceEnum": ["manual", "signal_report", "agent"], "TaskBootstrapRunSourceEnum": ["manual", "signal_report"], + # Completion providers are a subset of LLMProvider that excludes evaluation-only models. + "LLMCompletionProviderEnum": "products.ai_observability.backend.models.provider_keys.llm_completion_provider_choices", # # The same choice set is declared in more than one product. A shared Choices # class would cross a product boundary, so the entry names the set centrally. diff --git a/posthog/temporal/ai_observability/eval_reports/report_agent/prompts.py b/posthog/temporal/ai_observability/eval_reports/report_agent/prompts.py index 1a8430701fd4..94450ac4e60b 100644 --- a/posthog/temporal/ai_observability/eval_reports/report_agent/prompts.py +++ b/posthog/temporal/ai_observability/eval_reports/report_agent/prompts.py @@ -154,9 +154,14 @@ def build_eval_report_system_prompt( "- **`get_top_outcome_reasons(outcome, limit)`**: grouped reasoning strings for one outcome. " f"If omitted, outcome defaults to `{analysis_outcome}`.\n" ) - result_overview_detail = "Includes truncated reasoning." + result_overview_detail = "Includes truncated reasoning when the judge provides it." sample_ordering_signature = "" - sample_ordering_instruction = 'Rows include full reasoning. Use the default `order_by="recent"`.' + sample_ordering_instruction = ( + 'Rows include full reasoning when available. Use the default `order_by="recent"`. ' + "Some judges, including Jev, return no written reasoning. For those results, inspect the original " + "generation, trace, or session with the detail tools and ground your analysis in that source. " + "Do not invent a judge explanation or treat absent reasoning as an evaluation failure." + ) analysis_sample_arguments = f'outcome="{analysis_outcome}"' outcome_analysis_step = ( f"Inspect grouped reasons and sample relevant outcomes, using `{analysis_outcome}` and `{primary_outcome}` " diff --git a/posthog/temporal/ai_observability/evaluation_llm_judge.py b/posthog/temporal/ai_observability/evaluation_llm_judge.py index 6b34453ef95f..76386be60251 100644 --- a/posthog/temporal/ai_observability/evaluation_llm_judge.py +++ b/posthog/temporal/ai_observability/evaluation_llm_judge.py @@ -41,10 +41,13 @@ ModelNotFoundError, ModelPermissionError, ProviderConnectionError, + ProviderMismatchError, QuotaExceededError, RateLimitError, StructuredOutputParseError, ) +from products.ai_observability.backend.llm.types import CompletionResponse, Usage +from products.ai_observability.backend.llm.typesafe import TypeSafeClient, TypeSafeRateLimitError from products.ai_observability.backend.text_repr.formatters import add_line_numbers, reduce_by_uniform_sampling logger = structlog.get_logger(__name__) @@ -335,16 +338,55 @@ def call_llm_judge( capture_analytics=False, ) + probability: float | None = None try: - response = client.complete( - CompletionRequest( + if provider == "typesafe": + if provider_key is not None and provider_key.provider != provider: + raise ProviderMismatchError(provider_key.provider, provider) + typesafe_result = TypeSafeClient.evaluate_boolean( + api_key=provider_key.encrypted_config.get("api_key", "") if provider_key else "", model=model, - system=system_prompt, - messages=[{"role": "user", "content": user_prompt}], - provider=provider, - response_format=response_format, + prompt=evaluation["evaluation_config"]["prompt"], + source=user_prompt, + allows_na=allows_na, ) - ) + probability = typesafe_result.answers["verdict"].noul + applicable = not allows_na or typesafe_result.answers["applicable"].noul >= 0.5 + parsed: BooleanEvalResult | BooleanWithNAEvalResult = ( + BooleanWithNAEvalResult( + reasoning="", applicable=applicable, verdict=probability >= 0.5 if applicable else None + ) + if allows_na + else BooleanEvalResult(reasoning="", verdict=probability >= 0.5) + ) + model = typesafe_result.model + response = CompletionResponse( + content="", + model=model, + parsed=parsed, + usage=Usage( + input_tokens=typesafe_result.usage.input_tokens, + output_tokens=typesafe_result.usage.output_tokens, + total_tokens=typesafe_result.usage.input_tokens + typesafe_result.usage.output_tokens, + ), + ) + else: + response = client.complete( + CompletionRequest( + model=model, + system=system_prompt, + messages=[{"role": "user", "content": user_prompt}], + provider=provider, + response_format=response_format, + ) + ) + except TypeSafeRateLimitError as e: + increment_errors("rate_limit", provider=provider) + raise ApplicationError( + str(e), + {"error_type": "provider_unavailable", "provider": provider}, + next_retry_delay=timedelta(seconds=e.retry_after) if e.retry_after is not None else None, + ) from e except AuthenticationError: if is_byok: increment_user_errors("auth_error", provider=provider) @@ -496,4 +538,7 @@ def call_llm_judge( else: raise ValueError(f"Unexpected result type: {type(parsed_result)}") + if probability is not None: + result_dict["probability"] = probability + return result_dict diff --git a/posthog/temporal/ai_observability/evaluation_types.py b/posthog/temporal/ai_observability/evaluation_types.py index 5a35886967cb..d3b72f942f41 100644 --- a/posthog/temporal/ai_observability/evaluation_types.py +++ b/posthog/temporal/ai_observability/evaluation_types.py @@ -33,6 +33,7 @@ class EvaluationActivityResult(TypedDict, total=False): result_type: Required[Literal["boolean", "sentiment"]] reasoning: Required[str] verdict: NotRequired[bool | None] + probability: NotRequired[float] allows_na: NotRequired[bool] input_tokens: NotRequired[int] output_tokens: NotRequired[int] diff --git a/posthog/temporal/ai_observability/evaluation_workflow_activities.py b/posthog/temporal/ai_observability/evaluation_workflow_activities.py index 377271e1d712..c0ec2ceffc30 100644 --- a/posthog/temporal/ai_observability/evaluation_workflow_activities.py +++ b/posthog/temporal/ai_observability/evaluation_workflow_activities.py @@ -286,6 +286,8 @@ def build_evaluation_event_properties( properties["$ai_evaluation_provider"] = result.get("provider", "openai") properties["$ai_evaluation_key_type"] = "byok" if result.get("is_byok") else "posthog" properties["$ai_evaluation_key_id"] = result.get("key_id") + if "probability" in result: + properties["$ai_evaluation_probability"] = result["probability"] if result["result_type"] == "sentiment": if not result.get("skipped"): diff --git a/posthog/temporal/ai_observability/test_run_evaluation.py b/posthog/temporal/ai_observability/test_run_evaluation.py index 7b10a5f48fbb..601c77e6ae70 100644 --- a/posthog/temporal/ai_observability/test_run_evaluation.py +++ b/posthog/temporal/ai_observability/test_run_evaluation.py @@ -1,6 +1,6 @@ import json import uuid -from datetime import UTC, datetime +from datetime import UTC, datetime, timedelta from typing import Any import pytest @@ -42,8 +42,17 @@ status_reason_detail_for_terminal_user_error, terminal_user_error_result_from_application_error, ) -from .evaluation_llm_judge import JUDGE_EVENT_MAX_CHARS, TransientJudgeError, _execute_llm_judge_activity -from .evaluation_workflow_activities import LocalEvaluationOutcome, backfill_verdict_timestamp +from .evaluation_llm_judge import ( + JUDGE_EVENT_MAX_CHARS, + TransientJudgeError, + _execute_llm_judge_activity, + call_llm_judge, +) +from .evaluation_workflow_activities import ( + LocalEvaluationOutcome, + backfill_verdict_timestamp, + build_evaluation_event_properties, +) from .run_evaluation import ( BooleanEvalResult, BooleanWithNAEvalResult, @@ -75,6 +84,75 @@ def _mock_config_with_active_key(provider: str = "openai") -> MagicMock: return MagicMock(active_provider_key=key) +@pytest.mark.parametrize( + "probability,applicability,allows_na,verdict", + [(0.49, 1.0, False, False), (0.5, 1.0, False, True), (0.9, 0.1, True, None), (0.0, 0.9, True, False)], +) +def test_typesafe_judge_emits_boolean_probability_without_reasoning( + probability: float, applicability: float, allows_na: bool, verdict: bool | None +) -> None: + key = MagicMock(provider="typesafe", encrypted_config={"api_key": "test-typesafe-key"}) + resolved = MagicMock(provider="typesafe", model="jev-1.13.0", provider_key=key, is_byok=True) + response = MagicMock(status_code=200) + response.json.return_value = { + "model": "jev-1.13.0", + "answers": { + "verdict": {"type": "noul", "noul": probability}, + "applicable": {"type": "noul", "noul": applicability}, + }, + "usage": {"input_tokens": 120, "output_tokens": 10}, + } + evaluation = { + "id": "test-evaluation", + "name": "Politeness", + "team_id": 1, + "evaluation_config": {"prompt": "Is the response polite?"}, + } + with ( + patch("posthog.temporal.ai_observability.evaluation_llm_judge.model_spec") as spec, + patch("requests.request", return_value=response), + ): + spec.return_value.resolve.return_value = resolved + result = call_llm_judge( + evaluation=evaluation, + system_prompt="Unused generation instructions", + user_prompt="Hello!", + allows_na=allows_na, + ) + + assert result["verdict"] is verdict + assert result["reasoning"] == "" + assert result["probability"] == probability + assert result["total_tokens"] == 130 + if allows_na: + assert result["applicable"] is (applicability >= 0.5) + properties = build_evaluation_event_properties(evaluation, result, datetime.now(UTC)) + assert properties["$ai_evaluation_probability"] == probability + assert properties["$ai_model"] == "jev-1.13.0" + assert properties["$ai_evaluation_key_type"] == "byok" + + +def test_typesafe_rate_limit_retries_without_disabling_the_evaluation() -> None: + key = MagicMock(provider="typesafe", encrypted_config={"api_key": "test-typesafe-key"}) + with ( + patch("posthog.temporal.ai_observability.evaluation_llm_judge.model_spec") as spec, + patch("requests.request", return_value=MagicMock(status_code=429, headers={"Retry-After": "15"})), + pytest.raises(ApplicationError) as error, + ): + spec.return_value.resolve.return_value = MagicMock( + provider="typesafe", model="jev-1.13.0", provider_key=key, is_byok=True + ) + call_llm_judge( + evaluation={"team_id": 1, "evaluation_config": {"prompt": "Polite?"}}, + system_prompt="", + user_prompt="Hello!", + allows_na=False, + ) + assert not error.value.non_retryable + assert error.value.next_retry_delay == timedelta(seconds=15) + assert terminal_user_error_result_from_application_error(error.value, allows_na=False) is None + + def test_status_reason_detail_for_terminal_user_error_only_keeps_truncated_hog_errors(): hog_spec = require_user_error_spec("hog_error") permission_spec = require_user_error_spec("permission_error") diff --git a/products/ai_observability/backend/api/evaluation_config.py b/products/ai_observability/backend/api/evaluation_config.py index 4f658caf8da0..20f56270b836 100644 --- a/products/ai_observability/backend/api/evaluation_config.py +++ b/products/ai_observability/backend/api/evaluation_config.py @@ -14,7 +14,7 @@ from posthog.scopes import APIScopeObjectOrNotSupported from ..models.evaluation_config import EvaluationConfig -from ..models.provider_keys import LLMProviderKey +from ..models.provider_keys import LLMProvider, LLMProviderKey from .metrics import llma_track_latency from .provider_keys import LLMProviderKeySerializer @@ -109,6 +109,12 @@ def set_active_key(self, request: ValidatedRequest, **kwargs) -> Response: status=status.HTTP_404_NOT_FOUND, ) + if key.provider == LLMProvider.TYPESAFE: + return Response( + {"detail": "Select the TypeSafe key on an evaluation instead."}, + status=status.HTTP_400_BAD_REQUEST, + ) + if key.state != LLMProviderKey.State.OK: return Response( {"detail": f"Cannot activate key with state '{key.state}'. Please validate the key first."}, diff --git a/products/ai_observability/backend/api/evaluations.py b/products/ai_observability/backend/api/evaluations.py index dcbf33887400..ccc5c5b84856 100644 --- a/products/ai_observability/backend/api/evaluations.py +++ b/products/ai_observability/backend/api/evaluations.py @@ -43,6 +43,7 @@ from ..evaluation_conditions import build_condition_filter from ..hog import compile_ai_observability_hog from ..llm import DEFAULT_MODEL_BY_PROVIDER +from ..llm.typesafe import TypeSafeClient from ..models.evaluation_config import EvaluationConfig from ..models.evaluation_configs import ( EVALUATION_TEST_LOOKBACK_DAYS, @@ -251,6 +252,11 @@ def validate(self, data: dict[str, Any]) -> dict[str, Any]: errors = {field: "This field is required." for field in ("provider", "model") if field not in data} if errors: raise serializers.ValidationError(errors, code="required") + if data["provider"] == LLMProvider.TYPESAFE: + if data["model"] != TypeSafeClient.MODEL: + raise serializers.ValidationError({"model": "Select a supported Jev model."}) + if not data.get("provider_key_id"): + raise serializers.ValidationError({"provider_key_id": "Select a TypeSafe API key for this evaluation."}) return data def get_provider_key_name(self, obj: LLMModelConfiguration) -> str | None: diff --git a/products/ai_observability/backend/api/provider_keys.py b/products/ai_observability/backend/api/provider_keys.py index f5dae204f375..c8637d3608af 100644 --- a/products/ai_observability/backend/api/provider_keys.py +++ b/products/ai_observability/backend/api/provider_keys.py @@ -159,6 +159,10 @@ def validate(self, data): raise serializers.ValidationError({"api_key": "API key is required when creating a new provider key."}) provider = data.get("provider", getattr(self.instance, "provider", None)) + if self.instance is not None and provider != self.instance.provider: + raise serializers.ValidationError({"provider": "A key's provider cannot change. Create a new key instead."}) + if provider == LLMProvider.TYPESAFE and data.get("set_as_active"): + raise serializers.ValidationError({"set_as_active": "Select the TypeSafe key on an evaluation instead."}) if provider == LLMProvider.AZURE_OPENAI: has_endpoint = bool(data.get("azure_endpoint")) has_existing_endpoint = self.instance and self.instance.encrypted_config.get("azure_endpoint") diff --git a/products/ai_observability/backend/api/proxy.py b/products/ai_observability/backend/api/proxy.py index 37d8cc696a0f..f1f0d27dc1c6 100644 --- a/products/ai_observability/backend/api/proxy.py +++ b/products/ai_observability/backend/api/proxy.py @@ -49,7 +49,7 @@ get_playground_models, ) from products.ai_observability.backend.llm.errors import UnsupportedProviderError -from products.ai_observability.backend.models.provider_keys import LLMProvider, LLMProviderKey +from products.ai_observability.backend.models.provider_keys import LLMProviderKey, llm_completion_provider_choices from ee.hogai.utils.asgi import SyncIterableToAsync @@ -63,6 +63,7 @@ def models_cache_key(provider_key_id: str | uuid.UUID) -> str: PROVIDER_DISPLAY_NAMES: dict[str, str] = { + "typesafe": "TypeSafe", "openai": "OpenAI", "anthropic": "Anthropic", "gemini": "Gemini", @@ -78,7 +79,7 @@ class LLMProxyCompletionSerializer(serializers.Serializer): system = serializers.CharField(allow_blank=True) messages = serializers.ListField(child=serializers.DictField()) model = serializers.CharField() - provider = serializers.ChoiceField(choices=LLMProvider.choices) + provider = serializers.ChoiceField(choices=llm_completion_provider_choices()) thinking = serializers.BooleanField(default=False, required=False) temperature = serializers.FloatField(required=False) top_p = serializers.FloatField(required=False) diff --git a/products/ai_observability/backend/api/taggers.py b/products/ai_observability/backend/api/taggers.py index 27e599e15958..1af3bb8cf5eb 100644 --- a/products/ai_observability/backend/api/taggers.py +++ b/products/ai_observability/backend/api/taggers.py @@ -39,7 +39,7 @@ from ..hog import compile_ai_observability_hog from ..models.model_configuration import LLMModelConfiguration -from ..models.provider_keys import LLMProvider, LLMProviderKey +from ..models.provider_keys import LLMProviderKey, llm_completion_provider_choices from ..models.taggers import Tagger, TaggerType, validate_tagger_config from .metrics import llma_track_latency @@ -121,7 +121,9 @@ class TaggerConfigField(serializers.JSONField): class TaggerModelConfigurationWriteSerializer(serializers.Serializer): - provider = serializers.ChoiceField(choices=LLMProvider.choices, help_text="LLM provider to use for this tagger.") + provider = serializers.ChoiceField( + choices=llm_completion_provider_choices(), help_text="LLM provider to use for this tagger." + ) model = serializers.CharField(max_length=100, help_text="Provider model identifier to use for this tagger.") provider_key_id = serializers.UUIDField( required=False, diff --git a/products/ai_observability/backend/api/test/test_evaluation_config.py b/products/ai_observability/backend/api/test/test_evaluation_config.py index ec18af3aa0a9..ba3a43f60b8f 100644 --- a/products/ai_observability/backend/api/test/test_evaluation_config.py +++ b/products/ai_observability/backend/api/test/test_evaluation_config.py @@ -36,6 +36,20 @@ def _setup_team(): class TestEvaluationConfigViewSet(APIBaseTest): + def test_typesafe_cannot_become_the_shared_active_key(self) -> None: + key = LLMProviderKey.objects.create( + team=self.team, + provider="typesafe", + name="TypeSafe", + state="ok", + encrypted_config={"api_key": "test-typesafe-key"}, + ) + response = self.client.post( + f"/api/environments/{self.team.id}/llm_analytics/evaluation_config/set_active_key/", {"key_id": str(key.id)} + ) + self.assertEqual(response.status_code, status.HTTP_400_BAD_REQUEST) + self.assertFalse(EvaluationConfig.objects.filter(team=self.team, active_provider_key=key).exists()) + def test_unauthenticated_user_cannot_access_config(self): self.client.logout() response = self.client.get(f"/api/environments/{self.team.id}/llm_analytics/evaluation_config/") diff --git a/products/ai_observability/backend/api/test/test_evaluations.py b/products/ai_observability/backend/api/test/test_evaluations.py index cca163537cf8..004d55bc3ea4 100644 --- a/products/ai_observability/backend/api/test/test_evaluations.py +++ b/products/ai_observability/backend/api/test/test_evaluations.py @@ -54,6 +54,21 @@ def _setup_team(): class TestModelConfigurationSerializer(SimpleTestCase): + @parameterized.expand( + [ + ("missing_key", "jev-1.13.0", None, False), + ("unsupported_model", "other-model", str(uuid4()), False), + ("configured", "jev-1.13.0", str(uuid4()), True), + ] + ) + def test_typesafe_requires_supported_model_and_explicit_key( + self, _name: str, model: str, key_id: str | None, valid: bool + ) -> None: + serializer = ModelConfigurationSerializer( + data={"provider": "typesafe", "model": model, "provider_key_id": key_id} + ) + self.assertEqual(serializer.is_valid(), valid, serializer.errors) + @parameterized.expand( [ ("missing_provider", {"model": "gpt-5-mini"}, "provider"), @@ -114,16 +129,18 @@ def test_unauthenticated_user_cannot_access_evaluation_configs(self): response = self.client.get(f"/api/environments/{self.team.id}/evaluations/") self.assertEqual(response.status_code, status.HTTP_401_UNAUTHORIZED) - def test_can_create_evaluation_config(self): + @parameterized.expand([("openai", "gpt-5-mini"), ("typesafe", "jev-1.13.0")]) + def test_can_create_evaluation_config(self, provider: str, model: str) -> None: key = LLMProviderKey.objects.create( team=self.team, - provider="openai", + provider=provider, name="Active Key", state=LLMProviderKey.State.OK, encrypted_config={"api_key": "sk-test"}, created_by=self.user, ) - EvaluationConfig.objects.create(team=self.team, active_provider_key=key) + if provider == "openai": + EvaluationConfig.objects.create(team=self.team, active_provider_key=key) response = self.client.post( f"/api/environments/{self.team.id}/evaluations/", { @@ -131,7 +148,9 @@ def test_can_create_evaluation_config(self): "description": "Test Description", "enabled": True, "evaluation_type": "llm_judge", - "model_configuration": _DEFAULT_MODEL_CONFIGURATION, + "model_configuration": {"provider": provider, "model": model, "provider_key_id": str(key.id)} + if provider == "typesafe" + else _DEFAULT_MODEL_CONFIGURATION, "evaluation_config": {"prompt": "Test prompt"}, "output_type": "boolean", "output_config": {}, diff --git a/products/ai_observability/backend/api/test/test_provider_keys.py b/products/ai_observability/backend/api/test/test_provider_keys.py index 2a26070f7d54..f840c2af3209 100644 --- a/products/ai_observability/backend/api/test/test_provider_keys.py +++ b/products/ai_observability/backend/api/test/test_provider_keys.py @@ -1,19 +1,22 @@ from uuid import uuid4 from posthog.test.base import APIBaseTest -from unittest.mock import patch +from unittest.mock import Mock, patch from django.core.cache import cache +from django.test import SimpleTestCase from django.utils import timezone from parameterized import parameterized -from rest_framework import status +from rest_framework import serializers, status from posthog.constants import AvailableFeature from posthog.models import Organization, OrganizationMembership, Project, Team, User from products.access_control.backend.models.access_control import AccessControl -from products.ai_observability.backend.api.proxy import models_cache_key +from products.ai_observability.backend.api.provider_keys import LLMProviderKeySerializer +from products.ai_observability.backend.api.proxy import LLMProxyCompletionSerializer, models_cache_key +from products.ai_observability.backend.api.taggers import TaggerModelConfigurationWriteSerializer from products.ai_observability.backend.llm.providers.azure_openai import DEFAULT_API_VERSION from products.ai_observability.backend.models.evaluation_config import EvaluationConfig from products.ai_observability.backend.models.evaluations import Evaluation @@ -22,6 +25,36 @@ from products.ai_observability.backend.models.taggers import Tagger +class TestProviderKeySerializer(SimpleTestCase): + @parameterized.expand([(LLMProxyCompletionSerializer,), (TaggerModelConfigurationWriteSerializer,)]) + def test_typesafe_is_not_a_completion_provider(self, serializer_class: type[serializers.Serializer]) -> None: + serializer = serializer_class( + data={ + "provider": "typesafe", + "model": "jev-1.13.0", + "provider_key_id": str(uuid4()), + "system": "Reply politely.", + "messages": [{"role": "user", "content": "Hello!"}], + } + ) + self.assertFalse(serializer.is_valid()) + self.assertIn("provider", serializer.errors) + + @parameterized.expand([("openai", "typesafe"), ("typesafe", "openai")]) + def test_cannot_change_provider_of_an_existing_key(self, current: str, requested: str) -> None: + key = LLMProviderKey(provider=current, state="ok", encrypted_config={"api_key": "test-key"}) + serializer = LLMProviderKeySerializer(key, data={"provider": requested}, partial=True) + self.assertFalse(serializer.is_valid()) + self.assertIn("provider", serializer.errors) + + def test_typesafe_cannot_be_created_as_the_shared_key(self) -> None: + serializer = LLMProviderKeySerializer( + data={"provider": "typesafe", "name": "TypeSafe", "api_key": "test-key", "set_as_active": True} + ) + self.assertFalse(serializer.is_valid()) + self.assertIn("set_as_active", serializer.errors) + + def _setup_team(): org = Organization.objects.create(name="test") project = Project.objects.create(id=Team.objects.increment_id_sequence(), organization=org) @@ -85,13 +118,14 @@ def test_unauthenticated_user_cannot_access_provider_keys(self): response = self.client.get(f"/api/environments/{self.team.id}/llm_analytics/provider_keys/") self.assertEqual(response.status_code, status.HTTP_401_UNAUTHORIZED) + @parameterized.expand([("openai",), ("typesafe",)]) @patch("products.ai_observability.backend.api.provider_keys.validate_provider_key") - def test_can_create_provider_key(self, mock_validate): + def test_can_create_provider_key(self, provider: str, mock_validate: Mock) -> None: mock_validate.return_value = (LLMProviderKey.State.OK, None) response = self.client.post( f"/api/environments/{self.team.id}/llm_analytics/provider_keys/", - {"provider": "openai", "name": "My Key", "api_key": "sk-test-key-12345"}, + {"provider": provider, "name": "My Key", "api_key": "sk-test-key-12345"}, ) self.assertEqual(response.status_code, status.HTTP_201_CREATED) self.assertEqual(LLMProviderKey.objects.count(), 1) @@ -99,14 +133,14 @@ def test_can_create_provider_key(self, mock_validate): key = LLMProviderKey.objects.first() assert key is not None self.assertEqual(key.name, "My Key") - self.assertEqual(key.provider, "openai") + self.assertEqual(key.provider, provider) self.assertEqual(key.state, LLMProviderKey.State.OK) self.assertEqual(key.team, self.team) self.assertEqual(key.created_by, self.user) self.assertEqual(response.data["api_key_masked"], "sk-t...2345") self.assertNotIn("api_key", response.data) - mock_validate.assert_called_once_with("openai", "sk-test-key-12345") + mock_validate.assert_called_once_with(provider, "sk-test-key-12345") @patch("products.ai_observability.backend.api.provider_keys.validate_provider_key") def test_can_create_provider_key_with_set_as_active(self, mock_validate): diff --git a/products/ai_observability/backend/llm/client.py b/products/ai_observability/backend/llm/client.py index 8e496868a2cf..ad85f43a247a 100644 --- a/products/ai_observability/backend/llm/client.py +++ b/products/ai_observability/backend/llm/client.py @@ -15,6 +15,7 @@ CompletionResponse, StreamChunk, ) +from products.ai_observability.backend.llm.typesafe import TypeSafeClient if TYPE_CHECKING: from products.ai_observability.backend.llm.config import ProviderConfig @@ -81,16 +82,22 @@ def _resolve_credentials(self) -> tuple[str | None, str | None]: @classmethod def validate_key(cls, provider: str, api_key: str, **kwargs: Any) -> tuple[str, str | None]: """Validate an API key for a provider. Returns (state, error_message).""" + if provider == "typesafe": + return TypeSafeClient.validate_key(api_key) return _get_provider(provider).validate_key(api_key, **kwargs) @classmethod def list_models(cls, provider: str, api_key: str | None = None, **kwargs: Any) -> list[str]: """List available models for a provider.""" + if provider == "typesafe": + return [TypeSafeClient.MODEL] return _get_provider(provider).list_models(api_key, **kwargs) @classmethod def recommended_models(cls, provider: str) -> set[str]: """Return the set of curated/recommended model IDs for a provider.""" + if provider == "typesafe": + return {TypeSafeClient.MODEL} return _get_provider(provider).recommended_models() diff --git a/products/ai_observability/backend/llm/test/test_typesafe.py b/products/ai_observability/backend/llm/test/test_typesafe.py new file mode 100644 index 000000000000..7815eccd0188 --- /dev/null +++ b/products/ai_observability/backend/llm/test/test_typesafe.py @@ -0,0 +1,122 @@ +import pytest +from unittest.mock import Mock, patch + +from products.ai_observability.backend.llm.client import Client +from products.ai_observability.backend.llm.errors import ( + AuthenticationError, + ContextWindowExceededError, + ModelPermissionError, + ProviderConnectionError, + StructuredOutputParseError, +) +from products.ai_observability.backend.llm.typesafe import TypeSafeClient, TypeSafeRateLimitError + + +@pytest.mark.parametrize("status, expected_state", [(200, "ok"), (401, "invalid"), (403, "invalid"), (500, "error")]) +def test_typesafe_key_validation(status: int, expected_state: str) -> None: + response = Mock(status_code=status) + response.json.return_value = {"models": [{"name": "jev-latest"}]} + with patch("requests.request", return_value=response) as request: + state, message = Client.validate_key("typesafe", "test-typesafe-key") + + assert state == expected_state + assert (message is None) == (expected_state == "ok") + assert request.call_args.args == ("GET", "https://api.typesafe.ai/v1/models") + assert request.call_args.kwargs["headers"]["Authorization"] == "Bearer test-typesafe-key" + + +@pytest.mark.parametrize("allows_na", [False, True]) +def test_typesafe_boolean_request(allows_na: bool) -> None: + response = Mock(status_code=200) + response.json.return_value = { + "model": "jev-1.13.0", + "answers": {"verdict": {"type": "noul", "noul": 0.8}, "applicable": {"type": "noul", "noul": 0.2}}, + "usage": {"input_tokens": 120, "output_tokens": 10}, + } + with patch("requests.request", return_value=response) as request: + result = TypeSafeClient.evaluate_boolean( + api_key="test-typesafe-key", + model="jev-1.13.0", + prompt="Is the response polite?", + source="Hello!", + allows_na=allows_na, + ) + + assert result.answers["verdict"].noul == 0.8 + body = request.call_args.kwargs["json"] + assert body["state"] == "Hello!" + assert body["questions"]["verdict"] == {"type": "noul", "instructions": "Is the response polite?"} + assert ("applicable" in body["questions"]) == allows_na + assert request.call_args.kwargs["allow_redirects"] is False + + +@pytest.mark.parametrize("probability", [-0.1, 1.1, float("nan"), float("inf"), "0.8", True, None]) +def test_typesafe_rejects_invalid_probabilities(probability: object) -> None: + response = Mock(status_code=200) + response.json.return_value = { + "model": "jev-1.13.0", + "answers": {"verdict": {"type": "noul", "noul": probability}}, + "usage": {"input_tokens": 120, "output_tokens": 10}, + } + with patch("requests.request", return_value=response), pytest.raises(StructuredOutputParseError): + TypeSafeClient.evaluate_boolean( + api_key="test-typesafe-key", + model="jev-1.13.0", + prompt="Is the response polite?", + source="Hello!", + allows_na=False, + ) + + +@pytest.mark.parametrize("status", [429, 529]) +def test_typesafe_rate_limits_are_retryable(status: int) -> None: + response = Mock(status_code=status, headers={"Retry-After": "15"}) + with patch("requests.request", return_value=response), pytest.raises(TypeSafeRateLimitError) as error: + TypeSafeClient.evaluate_boolean( + api_key="test-typesafe-key", + model="jev-1.13.0", + prompt="Is the response polite?", + source="Hello!", + allows_na=False, + ) + assert error.value.retry_after == 15 + + +def test_typesafe_requires_a_key() -> None: + with patch("requests.request") as request, pytest.raises(AuthenticationError): + TypeSafeClient.evaluate_boolean( + api_key="", model="jev-1.13.0", prompt="Polite?", source="Hello!", allows_na=False + ) + request.assert_not_called() + + +@pytest.mark.parametrize("answers", [{}, {"verdict": {"type": "noul", "noul": 0.9}}]) +def test_typesafe_requires_every_requested_answer(answers: dict[str, object]) -> None: + response = Mock(status_code=200) + response.json.return_value = { + "model": "jev-1.13.0", + "answers": answers, + "usage": {"input_tokens": 12, "output_tokens": 2}, + } + with patch("requests.request", return_value=response), pytest.raises(StructuredOutputParseError): + TypeSafeClient.evaluate_boolean( + api_key="test-typesafe-key", model="jev-1.13.0", prompt="Polite?", source="Hello!", allows_na=True + ) + + +@pytest.mark.parametrize( + "status,message,error_type", + [ + (401, "Invalid key", AuthenticationError), + (403, "Access denied", ModelPermissionError), + (500, "Unavailable", ProviderConnectionError), + (422, "Input exceeds the context window", ContextWindowExceededError), + (422, "Invalid question", StructuredOutputParseError), + ], +) +def test_typesafe_preserves_error_categories(status: int, message: str, error_type: type[Exception]) -> None: + response = Mock(status_code=status, text=message) + with patch("requests.request", return_value=response), pytest.raises(error_type): + TypeSafeClient.evaluate_boolean( + api_key="test-typesafe-key", model="jev-1.13.0", prompt="Polite?", source="Hello!", allows_na=False + ) diff --git a/products/ai_observability/backend/llm/typesafe.py b/products/ai_observability/backend/llm/typesafe.py new file mode 100644 index 000000000000..a005e3389fab --- /dev/null +++ b/products/ai_observability/backend/llm/typesafe.py @@ -0,0 +1,120 @@ +import math +from datetime import UTC, datetime +from email.utils import parsedate_to_datetime +from typing import Literal + +import requests +from pydantic import BaseModel, Field, ValidationError + +from products.ai_observability.backend.llm.errors import ( + AuthenticationError, + ContextWindowExceededError, + ModelNotFoundError, + ModelPermissionError, + ProviderConnectionError, + RateLimitError, + StructuredOutputParseError, + is_context_window_error_message, +) + + +class NoulAnswer(BaseModel): + type: Literal["noul"] + noul: float = Field(strict=True, ge=0, le=1, allow_inf_nan=False) + + +class TypeSafeUsage(BaseModel): + input_tokens: int = Field(strict=True, ge=0) + output_tokens: int = Field(strict=True, ge=0) + + +class TypeSafeResponse(BaseModel): + model: str = Field(min_length=1) + answers: dict[str, NoulAnswer] + usage: TypeSafeUsage + + +class TypeSafeRateLimitError(RateLimitError): + def __init__(self, retry_after: str | None) -> None: + super().__init__("TypeSafe is temporarily rate limiting requests.") + self.retry_after: float | None = None + if retry_after: + try: + delay = float(retry_after) + except ValueError: + try: + delay = (parsedate_to_datetime(retry_after) - datetime.now(UTC)).total_seconds() + except (ValueError, TypeError, OverflowError): + return + if math.isfinite(delay): + self.retry_after = max(1, min(delay, 300)) + + +class TypeSafeClient: + MODEL = "jev-1.13.0" + + @staticmethod + def evaluate_boolean(*, api_key: str, model: str, prompt: str, source: str, allows_na: bool) -> TypeSafeResponse: + if not api_key: + raise AuthenticationError("A TypeSafe API key is required.") + questions = {"verdict": {"type": "noul", "instructions": prompt}} + if allows_na: + questions["applicable"] = { + "type": "noul", + "instructions": ( + "Do these evaluation criteria apply to this input? Answer true when the criteria can be " + "evaluated, even if they are not met. Answer false only when they are not relevant.\n\n" + prompt + ), + } + try: + response = requests.request( + "POST", + "https://api.typesafe.ai/v1/systemone", + headers={"Authorization": f"Bearer {api_key}"}, + json={"model": model, "state": source, "questions": questions}, + timeout=60, + allow_redirects=False, + ) + except requests.RequestException as error: + raise ProviderConnectionError("Could not reach TypeSafe.") from error + + if response.status_code == 401: + raise AuthenticationError("TypeSafe rejected this API key.") + if response.status_code == 403: + raise ModelPermissionError(model) + if response.status_code == 404: + raise ModelNotFoundError(model) + if response.status_code in (429, 529): + raise TypeSafeRateLimitError(response.headers.get("Retry-After")) + if response.status_code >= 500: + raise ProviderConnectionError("TypeSafe is temporarily unavailable.") + if response.status_code == 422 and is_context_window_error_message(response.text): + raise ContextWindowExceededError("This input exceeds Jev's context window.") + if response.status_code != 200: + raise StructuredOutputParseError("TypeSafe rejected the evaluation request. Check the model and criteria.") + try: + result = TypeSafeResponse.model_validate(response.json()) + except (ValidationError, ValueError) as error: + raise StructuredOutputParseError("TypeSafe returned an invalid evaluation response.") from error + # boffin: Missing answers cannot become a false verdict. + if not questions.keys() <= result.answers.keys(): + raise StructuredOutputParseError("TypeSafe did not answer every evaluation question.") + return result + + @staticmethod + def validate_key(api_key: str) -> tuple[str, str | None]: + try: + response = requests.request( + "GET", + "https://api.typesafe.ai/v1/models", + headers={"Authorization": f"Bearer {api_key}"}, + timeout=30, + allow_redirects=False, + ) + except requests.RequestException: + return "error", "Could not reach TypeSafe. Try again." + if response.status_code == 200: + return "ok", None + if response.status_code in (401, 403): + return "invalid", "TypeSafe rejected this API key. Check the key and try again." + return "error", "Could not validate the TypeSafe key. Try again." diff --git a/products/ai_observability/backend/models/provider_keys.py b/products/ai_observability/backend/models/provider_keys.py index b98abb31da22..b51f27473086 100644 --- a/products/ai_observability/backend/models/provider_keys.py +++ b/products/ai_observability/backend/models/provider_keys.py @@ -22,6 +22,7 @@ class LLMProvider(models.TextChoices): TOGETHER_AI = "together_ai", "Together AI" MINIMAX = "minimax", "MiniMax" ZEABUR = "zeabur", "Zeabur AI Hub" + TYPESAFE = "typesafe", "TypeSafe" def llm_provider_choices() -> list[tuple[str, str | Promise]]: @@ -29,6 +30,10 @@ def llm_provider_choices() -> list[tuple[str, str | Promise]]: return list(LLMProvider.choices) +def llm_completion_provider_choices() -> list[tuple[str, str | Promise]]: + return [(provider, label) for provider, label in LLMProvider.choices if provider != LLMProvider.TYPESAFE] + + class LLMProviderKey(UUIDTModel): class State(models.TextChoices): UNKNOWN = "unknown" diff --git a/products/ai_observability/frontend/ByokModelPickerNotice.stories.tsx b/products/ai_observability/frontend/ByokModelPickerNotice.stories.tsx index ae66fddb9b58..a16df5161d64 100644 --- a/products/ai_observability/frontend/ByokModelPickerNotice.stories.tsx +++ b/products/ai_observability/frontend/ByokModelPickerNotice.stories.tsx @@ -11,49 +11,72 @@ import { LLMProviderKey, LLMProviderKeyState } from './settings/llmProviderKeysL interface StoryArgs { keyState: LLMProviderKeyState | null modelsFail: boolean + typesafe?: boolean } function providerKey(state: LLMProviderKeyState): Partial { return { id: 'key-1', provider: 'openai', name: 'Production', state } } -function PickerWithNotice(): JSX.Element { - const { providerModelGroups, hasByokKeys, byokModelsLoading, providerKeysLoading } = useValues(modelPickerLogic) +function PickerWithNotice({ typesafe = false }: { typesafe?: boolean }): JSX.Element { + const { providerModelGroups, evaluationProviderModelGroups, hasByokKeys, byokModelsLoading, providerKeysLoading } = + useValues(modelPickerLogic) return ( // The story root has no width of its own, so the picker column is sized here to match the form it sits in. -
+
{}} - groups={providerModelGroups} + groups={typesafe ? evaluationProviderModelGroups : providerModelGroups} loading={byokModelsLoading || providerKeysLoading} - footerLink={getModelPickerFooterLink(hasByokKeys)} + footerLink={getModelPickerFooterLink( + typesafe ? evaluationProviderModelGroups.some((group) => !group.disabledReason) : hasByokKeys + )} /> - +
) } const meta: Meta = { title: 'Scenes-App/AI observability/BYOK model picker notice', - render: ({ keyState, modelsFail }) => { + render: ({ keyState, modelsFail, typesafe }) => { useStorybookMocks({ get: { '/api/environments/:team_id/llm_analytics/provider_keys/': { - results: keyState ? [providerKey(keyState)] : [], + results: keyState + ? [ + { + ...providerKey(keyState), + ...(typesafe ? { provider: 'typesafe', name: 'TypeSafe' } : {}), + }, + ] + : [], }, '/api/environments/:team_id/llm_analytics/evaluation_config/': { active_provider_key: null }, // Only the per-key request fails. The playground list uses the same path without a key id. '/api/llm_proxy/models/': ({ request }) => modelsFail && new URL(request.url).searchParams.get('provider_key_id') ? [500, { error: 'Internal error' }] - : [200, []], + : [ + 200, + typesafe + ? [ + { + id: 'jev-1.13.0', + name: 'jev-1.13.0', + provider: 'TypeSafe', + is_recommended: true, + }, + ] + : [], + ], }, }) - return + return }, } export default meta @@ -69,3 +92,7 @@ export const NoUsableProviderKeys: StoryObj = { export const ModelsFailedToLoad: StoryObj = { args: { keyState: 'ok', modelsFail: true }, } + +export const Jev: StoryObj = { + args: { keyState: 'ok', modelsFail: false, typesafe: true }, +} diff --git a/products/ai_observability/frontend/ByokModelPickerNotice.tsx b/products/ai_observability/frontend/ByokModelPickerNotice.tsx index 0be94b7ed9be..2db1153ae964 100644 --- a/products/ai_observability/frontend/ByokModelPickerNotice.tsx +++ b/products/ai_observability/frontend/ByokModelPickerNotice.tsx @@ -7,8 +7,9 @@ import { modelPickerLogic } from './modelPickerLogic' import { providerLabel } from './settings/providerKeyStateUtils' /** Explains an empty or short model picker on the surfaces that only run on the team's own provider keys. */ -export function ByokModelPickerNotice(): JSX.Element | null { - const { byokModelNotice } = useValues(modelPickerLogic) +export function ByokModelPickerNotice({ forEvaluation = false }: { forEvaluation?: boolean }): JSX.Element | null { + const { byokModelNotice: generativeModelNotice, evaluationModelNotice } = useValues(modelPickerLogic) + const byokModelNotice = forEvaluation ? evaluationModelNotice : generativeModelNotice const { loadByokModels } = useActions(modelPickerLogic) switch (byokModelNotice?.kind) { diff --git a/products/ai_observability/frontend/ConversationDisplay/EvaluationDisplay.tsx b/products/ai_observability/frontend/ConversationDisplay/EvaluationDisplay.tsx index f614e648e94c..fdb90285bfff 100644 --- a/products/ai_observability/frontend/ConversationDisplay/EvaluationDisplay.tsx +++ b/products/ai_observability/frontend/ConversationDisplay/EvaluationDisplay.tsx @@ -7,14 +7,16 @@ import { urls } from 'scenes/urls' import { EventType } from '~/types' +import { EvaluationExplanation } from '../components/EvaluationExplanation' import { EvaluationResultTag } from '../components/EvaluationResultTag' import { MetadataTag } from '../components/MetadataTag' import { llmEvaluationsLogic } from '../evaluations/llmEvaluationsLogic' -import { normalizeEvaluationResultProperties } from '../utils' +import { normalizeEvaluationResultProperties, normalizeOptionalNumber } from '../utils' export function EvaluationDisplay({ eventProperties }: { eventProperties: EventType['properties'] }): JSX.Element { const { detectorEvaluationIds } = useValues(llmEvaluationsLogic) const reasoning = eventProperties.$ai_evaluation_reasoning + const probability = normalizeOptionalNumber(eventProperties.$ai_evaluation_probability) const evaluationName = eventProperties.$ai_evaluation_name const model = eventProperties.$ai_model ?? eventProperties.$ai_evaluation_model const traceId = eventProperties.$ai_trace_id @@ -55,10 +57,14 @@ export function EvaluationDisplay({ eventProperties }: { eventProperties: EventT )}
- {reasoning && ( + {(reasoning || probability !== null) && (
-
REASONING
-
{reasoning}
+
Evaluation details
+ {probability !== null ? ( + + ) : ( +
{reasoning}
+ )}
)} diff --git a/products/ai_observability/frontend/components/EvaluationExplanation.stories.tsx b/products/ai_observability/frontend/components/EvaluationExplanation.stories.tsx new file mode 100644 index 000000000000..308817da1549 --- /dev/null +++ b/products/ai_observability/frontend/components/EvaluationExplanation.stories.tsx @@ -0,0 +1,21 @@ +import type { Meta, StoryObj } from '@storybook/react' + +import { EvaluationExplanation } from './EvaluationExplanation' + +const meta: Meta = { + title: 'Scenes-App/AI observability/Evaluation explanation', + component: EvaluationExplanation, +} +export default meta + +export const Reasoning: StoryObj = { + args: { reasoning: 'The response answers the question and includes a polite greeting.' }, +} + +export const Jev: StoryObj = { + args: { probability: 0.84 }, +} + +export const JevZeroProbability: StoryObj = { + args: { probability: 0 }, +} diff --git a/products/ai_observability/frontend/components/EvaluationExplanation.tsx b/products/ai_observability/frontend/components/EvaluationExplanation.tsx new file mode 100644 index 000000000000..1a6262e677d4 --- /dev/null +++ b/products/ai_observability/frontend/components/EvaluationExplanation.tsx @@ -0,0 +1,24 @@ +import { Tooltip } from '@posthog/lemon-ui' + +export function EvaluationExplanation({ + reasoning, + probability, +}: { + reasoning?: string + probability?: number | null +}): JSX.Element { + if (probability != null) { + return ( + + {`${(probability * 100).toFixed(1)}% probability of true`} + + ) + } + return ( + +
+
{reasoning || 'No reasoning provided'}
+
+
+ ) +} diff --git a/products/ai_observability/frontend/components/GenerationEvalRunsTable.tsx b/products/ai_observability/frontend/components/GenerationEvalRunsTable.tsx index 2c97a54e5291..6afc9894c5da 100644 --- a/products/ai_observability/frontend/components/GenerationEvalRunsTable.tsx +++ b/products/ai_observability/frontend/components/GenerationEvalRunsTable.tsx @@ -1,6 +1,6 @@ import { BuiltLogic, useValues } from 'kea' -import { LemonTable, Link, Tooltip } from '@posthog/lemon-ui' +import { LemonTable, Link } from '@posthog/lemon-ui' import { TZLabel } from 'lib/components/TZLabel' import { LemonTableColumns } from 'lib/lemon-ui/LemonTable' @@ -9,6 +9,7 @@ import { urls } from 'scenes/urls' import { llmEvaluationsLogic } from '../evaluations/llmEvaluationsLogic' import { EvaluationRun } from '../evaluations/types' import type { generationEvaluationRunsLogicType } from '../generationEvaluationRunsLogic' +import { EvaluationExplanation } from './EvaluationExplanation' import { EvaluationResultTag, getEvaluationResultSortValue } from './EvaluationResultTag' import { EvaluationRunTargetCell } from './EvaluationRunTargetCell' @@ -59,15 +60,9 @@ export function GenerationEvalRunsTable({ }, }, { - title: 'Reasoning', + title: 'Details', key: 'reasoning', - render: (_, run) => ( - -
-
{run.reasoning}
-
-
- ), + render: (_, run) => , }, ] diff --git a/products/ai_observability/frontend/evaluations/AIObservabilityEvaluation.tsx b/products/ai_observability/frontend/evaluations/AIObservabilityEvaluation.tsx index 9137d65d1e60..2d3a05d85ad1 100644 --- a/products/ai_observability/frontend/evaluations/AIObservabilityEvaluation.tsx +++ b/products/ai_observability/frontend/evaluations/AIObservabilityEvaluation.tsx @@ -126,7 +126,9 @@ export function AIObservabilityEvaluation(): JSX.Element { return } const openInPlaygroundUrl = - evaluationTypeUsesModelConfiguration(evaluation.evaluation_type) && evaluation.id + evaluationTypeUsesModelConfiguration(evaluation.evaluation_type) && + evaluation.id && + evaluation.model_configuration?.provider !== 'typesafe' ? combineUrl(urls.aiObservabilityPlayground(), { source_evaluation_id: evaluation.id }).url : null @@ -903,17 +905,18 @@ export function AIObservabilityEvaluation(): JSX.Element { } function EvaluationModelPicker(): JSX.Element { - const { hasByokKeys, byokModels, providerModelGroups, byokModelsLoading, providerKeysLoading } = + const { byokModels, evaluationProviderModelGroups, byokModelsLoading, providerKeysLoading } = useValues(modelPickerLogic) - const { selectedModel, selectedPickerProviderKeyId, modelSelectionRequired } = useValues(llmEvaluationLogic) + const { selectedModel, selectedPickerProviderKeyId, modelSelectionRequired, evaluation } = + useValues(llmEvaluationLogic) const { selectModelFromPicker } = useActions(llmEvaluationLogic) // Evals always run on the team's own provider key, so only BYOK models are offered. const selectedModelName = byokModels.find((m) => m.id === selectedModel)?.name - const groups = providerModelGroups + const groups = evaluationProviderModelGroups const loading = byokModelsLoading || providerKeysLoading - const footerLink = getModelPickerFooterLink(hasByokKeys) + const footerLink = getModelPickerFooterLink(groups.some((group) => !group.disabledReason)) return (
@@ -935,7 +938,13 @@ function EvaluationModelPicker(): JSX.Element { selectedModelName={selectedModelName} data-attr="evaluation-model-selector" /> - + + {evaluation?.model_configuration?.provider === 'typesafe' && ( +

+ Jev returns a probability without written reasoning. A probability of 50% or higher + produces a true result. +

+ )} {modelSelectionRequired && !selectedModel && (

Select a judge model.

)} diff --git a/products/ai_observability/frontend/evaluations/components/EvaluationRunsTable.tsx b/products/ai_observability/frontend/evaluations/components/EvaluationRunsTable.tsx index 4c0ba8361983..aa1b938395d6 100644 --- a/products/ai_observability/frontend/evaluations/components/EvaluationRunsTable.tsx +++ b/products/ai_observability/frontend/evaluations/components/EvaluationRunsTable.tsx @@ -1,11 +1,12 @@ import { useActions, useValues } from 'kea' import { IconRefresh } from '@posthog/icons' -import { LemonBanner, LemonButton, LemonSegmentedButton, LemonTable, LemonTag, Tooltip } from '@posthog/lemon-ui' +import { LemonBanner, LemonButton, LemonSegmentedButton, LemonTable, LemonTag } from '@posthog/lemon-ui' import { TZLabel } from 'lib/components/TZLabel' import { LemonTableColumns } from 'lib/lemon-ui/LemonTable' +import { EvaluationExplanation } from '../../components/EvaluationExplanation' import { EvaluationResultTag, getEvaluationResultSortValue } from '../../components/EvaluationResultTag' import { EvaluationRunTargetCell } from '../../components/EvaluationRunTargetCell' import { evaluationIsDetector } from '../constants' @@ -107,15 +108,9 @@ export function EvaluationRunsTable(): JSX.Element { }, }, { - title: 'Reasoning', + title: 'Details', key: 'reasoning', - render: (_, run) => ( - -
-
{run.reasoning}
-
-
- ), + render: (_, run) => , }, { title: 'Status', diff --git a/products/ai_observability/frontend/evaluations/types.ts b/products/ai_observability/frontend/evaluations/types.ts index 35aa972762b8..d92281b64eea 100644 --- a/products/ai_observability/frontend/evaluations/types.ts +++ b/products/ai_observability/frontend/evaluations/types.ts @@ -143,6 +143,7 @@ export interface EvaluationRun { // evaluation disallows N/A, so it has to be read alongside this rather than on its own. skipped?: boolean reasoning: string + probability?: number | null status: 'completed' | 'failed' | 'running' } diff --git a/products/ai_observability/frontend/generated/api.schemas.ts b/products/ai_observability/frontend/generated/api.schemas.ts index ca36ebd4e839..7d5dd381ca0d 100644 --- a/products/ai_observability/frontend/generated/api.schemas.ts +++ b/products/ai_observability/frontend/generated/api.schemas.ts @@ -850,6 +850,7 @@ export const EvaluationTargetEnumApi = { * * `together_ai` - Together AI * * `minimax` - MiniMax * * `zeabur` - Zeabur AI Hub + * * `typesafe` - TypeSafe */ export type LLMProviderEnumApi = (typeof LLMProviderEnumApi)[keyof typeof LLMProviderEnumApi] @@ -863,6 +864,7 @@ export const LLMProviderEnumApi = { TogetherAi: 'together_ai', Minimax: 'minimax', Zeabur: 'zeabur', + Typesafe: 'typesafe', } as const /** @@ -3041,6 +3043,32 @@ export interface TaggerConditionApi { properties?: TaggerConditionApiPropertiesItem[] } +/** + * * `openai` - Openai + * * `anthropic` - Anthropic + * * `gemini` - Gemini + * * `openrouter` - Openrouter + * * `fireworks` - Fireworks + * * `azure_openai` - Azure OpenAI + * * `together_ai` - Together AI + * * `minimax` - MiniMax + * * `zeabur` - Zeabur AI Hub + */ +export type LLMCompletionProviderEnumApi = + (typeof LLMCompletionProviderEnumApi)[keyof typeof LLMCompletionProviderEnumApi] + +export const LLMCompletionProviderEnumApi = { + Openai: 'openai', + Anthropic: 'anthropic', + Gemini: 'gemini', + Openrouter: 'openrouter', + Fireworks: 'fireworks', + AzureOpenai: 'azure_openai', + TogetherAi: 'together_ai', + Minimax: 'minimax', + Zeabur: 'zeabur', +} as const + /** * Nested serializer for model configuration. */ @@ -3056,7 +3084,7 @@ export interface TaggerModelConfigurationApi { * * `together_ai` - Together AI * * `minimax` - MiniMax * * `zeabur` - Zeabur AI Hub */ - provider: LLMProviderEnumApi + provider: LLMCompletionProviderEnumApi /** * Provider model identifier to use for this tagger. * @maxLength 100 @@ -3110,7 +3138,7 @@ export interface TaggerModelConfigurationWriteApi { * * `together_ai` - Together AI * * `minimax` - MiniMax * * `zeabur` - Zeabur AI Hub */ - provider: LLMProviderEnumApi + provider: LLMCompletionProviderEnumApi /** * Provider model identifier to use for this tagger. * @maxLength 100 @@ -3537,6 +3565,7 @@ export const LlmAnalyticsModelsRetrieveProvider = { Openai: 'openai', Openrouter: 'openrouter', TogetherAi: 'together_ai', + Typesafe: 'typesafe', Zeabur: 'zeabur', } as const diff --git a/products/ai_observability/frontend/generated/api.zod.ts b/products/ai_observability/frontend/generated/api.zod.ts index 4e21ad828c8c..5fab62024e2b 100644 --- a/products/ai_observability/frontend/generated/api.zod.ts +++ b/products/ai_observability/frontend/generated/api.zod.ts @@ -438,9 +438,10 @@ export const EvaluationsCreateBody = /* @__PURE__ */ zod 'together_ai', 'minimax', 'zeabur', + 'typesafe', ]) .describe( - '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub' + '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub\n\* `typesafe` - TypeSafe' ), model: zod.string().max(evaluationsCreateBodyModelConfigurationOneModelMax), provider_key_id: zod @@ -742,9 +743,10 @@ export const EvaluationsUpdateBody = /* @__PURE__ */ zod 'together_ai', 'minimax', 'zeabur', + 'typesafe', ]) .describe( - '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub' + '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub\n\* `typesafe` - TypeSafe' ), model: zod.string().max(evaluationsUpdateBodyModelConfigurationOneModelMax), provider_key_id: zod @@ -950,9 +952,10 @@ export const EvaluationsPartialUpdateBody = /* @__PURE__ */ zod 'together_ai', 'minimax', 'zeabur', + 'typesafe', ]) .describe( - '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub' + '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub\n\* `typesafe` - TypeSafe' ), model: zod.string().max(evaluationsPartialUpdateBodyModelConfigurationOneModelMax), provider_key_id: zod @@ -1530,9 +1533,10 @@ export const LlmAnalyticsProviderKeysCreateBody = /* @__PURE__ */ zod.object({ 'together_ai', 'minimax', 'zeabur', + 'typesafe', ]) .describe( - '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub' + '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub\n\* `typesafe` - TypeSafe' ), name: zod.string().max(llmAnalyticsProviderKeysCreateBodyNameMax), api_key: zod.string().optional(), @@ -1563,9 +1567,10 @@ export const LlmAnalyticsProviderKeysUpdateBody = /* @__PURE__ */ zod.object({ 'together_ai', 'minimax', 'zeabur', + 'typesafe', ]) .describe( - '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub' + '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub\n\* `typesafe` - TypeSafe' ), name: zod.string().max(llmAnalyticsProviderKeysUpdateBodyNameMax), api_key: zod.string().optional(), @@ -1596,10 +1601,11 @@ export const LlmAnalyticsProviderKeysPartialUpdateBody = /* @__PURE__ */ zod.obj 'together_ai', 'minimax', 'zeabur', + 'typesafe', ]) .optional() .describe( - '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub' + '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub\n\* `typesafe` - TypeSafe' ), name: zod.string().max(llmAnalyticsProviderKeysPartialUpdateBodyNameMax).optional(), api_key: zod.string().optional(), @@ -1630,9 +1636,10 @@ export const LlmAnalyticsProviderKeysValidateCreateBody = /* @__PURE__ */ zod.ob 'together_ai', 'minimax', 'zeabur', + 'typesafe', ]) .describe( - '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub' + '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub\n\* `typesafe` - TypeSafe' ), name: zod.string().max(llmAnalyticsProviderKeysValidateCreateBodyNameMax), api_key: zod.string().optional(), diff --git a/products/ai_observability/frontend/modelPickerLogic.test.ts b/products/ai_observability/frontend/modelPickerLogic.test.ts index 2fbf369d3d21..ede2501f8805 100644 --- a/products/ai_observability/frontend/modelPickerLogic.test.ts +++ b/products/ai_observability/frontend/modelPickerLogic.test.ts @@ -45,6 +45,40 @@ describe('modelPickerLogic', () => { }) describe('loadByokModels', () => { + it('offers Jev only to evaluation model pickers', async () => { + useMocks({ + get: { + '/api/environments/:team_id/llm_analytics/provider_keys/': { + results: [{ id: 'key-typesafe', provider: 'typesafe', name: 'TypeSafe', state: 'ok' }], + }, + '/api/environments/:team_id/llm_analytics/evaluation_config/': { active_provider_key: null }, + '/api/llm_proxy/models/': ({ request }) => + new URL(request.url).searchParams.get('provider_key_id') + ? [ + 200, + [ + { + id: 'jev-1.13.0', + name: 'jev-1.13.0', + provider: 'TypeSafe', + is_recommended: true, + }, + ], + ] + : [200, []], + }, + }) + logic = modelPickerLogic() + logic.mount() + await expectLogic(logic).toFinishAllListeners() + + expect(logic.values.evaluationProviderModelGroups[0].models[0].id).toBe('jev-1.13.0') + expect(logic.values.providerModelGroups).toEqual([]) + expect(logic.values.generativeByokModels).toEqual([]) + expect(logic.values.hasByokKeys).toBe(false) + expect(logic.values.evaluationModelNotice).toBeNull() + }) + it('should load and attach providerKeyId to models from valid keys', async () => { useMocks({ get: { diff --git a/products/ai_observability/frontend/modelPickerLogic.ts b/products/ai_observability/frontend/modelPickerLogic.ts index 841a7e9ef124..9f1da5bf9c08 100644 --- a/products/ai_observability/frontend/modelPickerLogic.ts +++ b/products/ai_observability/frontend/modelPickerLogic.ts @@ -42,6 +42,26 @@ const UNHEALTHY_KEY_REASON = 'This provider key has an issue. Check your provide const UNAVAILABLE_KEY_REASON = "Couldn't load models for this key. Try again in a moment." const UNAVAILABLE_KEY_SUFFIX = ' (Unavailable)' +function getByokModelNotice( + providerKeys: LLMProviderKey[], + providerKeysLoading: boolean, + byokModelsLoading: boolean, + failedByokProviderKeyIds: string[], + providerModelGroups: ProviderModelGroup[] +): ByokModelNotice | null { + if (providerKeysLoading || byokModelsLoading) { + return null + } + const failedKeys = providerKeys.filter((key) => failedByokProviderKeyIds.includes(key.id)) + if (failedKeys.length > 0) { + return { kind: 'models-failed', keys: failedKeys } + } + if (providerModelGroups.some((group) => !group.disabledReason)) { + return null + } + return providerKeys.length === 0 ? { kind: 'no-keys' } : { kind: 'no-usable-keys' } +} + function providerKeyGroupLabel(key: LLMProviderKey, keysPerProvider: Record, suffix = ''): string { const label = providerLabel(key.provider) return (keysPerProvider[key.provider] ?? 0) > 1 ? `${label} (${key.name})${suffix}` : `${label}${suffix}` @@ -94,7 +114,10 @@ export interface modelPickerLogicValues { byokModelNotice: ByokModelNotice | null byokModels: ModelOption[] byokModelsLoading: boolean + evaluationModelNotice: ByokModelNotice | null + evaluationProviderModelGroups: ProviderModelGroup[] failedByokProviderKeyIds: string[] + generativeByokModels: ModelOption[] hasByokKeys: boolean playgroundModels: ModelOption[] playgroundModelsLoading: boolean @@ -178,12 +201,21 @@ export interface modelPickerLogicActions { export interface modelPickerLogicMeta { __keaTypeGenInternalSelectorTypes: { hasByokKeys: (providerKeys: LLMProviderKey[]) => boolean + generativeByokModels: (byokModels: ModelOption[], providerKeys: LLMProviderKey[]) => ModelOption[] playgroundProviderModelGroups: (playgroundModels: ModelOption[]) => ProviderModelGroup[] - providerModelGroups: ( + evaluationProviderModelGroups: ( byokModels: ModelOption[], providerKeys: LLMProviderKey[], failedByokProviderKeyIds: string[] ) => ProviderModelGroup[] + providerModelGroups: (evaluationProviderModelGroups: ProviderModelGroup[]) => ProviderModelGroup[] + evaluationModelNotice: ( + providerKeys: LLMProviderKey[], + providerKeysLoading: boolean, + byokModelsLoading: boolean, + failedByokProviderKeyIds: string[], + evaluationProviderModelGroups: ProviderModelGroup[] + ) => ByokModelNotice | null byokModelNotice: ( providerKeys: LLMProviderKey[], providerKeysLoading: boolean, @@ -314,14 +346,22 @@ export const modelPickerLogic = kea([ selectors({ hasByokKeys: [ (s) => [s.providerKeys], - (providerKeys: LLMProviderKey[]): boolean => providerKeys.some((k) => k.state === 'ok'), + (providerKeys: LLMProviderKey[]): boolean => + providerKeys.some((k) => k.state === 'ok' && k.provider !== 'typesafe'), + ], + generativeByokModels: [ + (s) => [s.byokModels, s.providerKeys], + (models: ModelOption[], keys: LLMProviderKey[]): ModelOption[] => + models.filter( + (model) => !keys.some((key) => key.id === model.providerKeyId && key.provider === 'typesafe') + ), ], playgroundProviderModelGroups: [ (s) => [s.playgroundModels], (playgroundModels: ModelOption[]): ProviderModelGroup[] => buildPlaygroundProviderModelGroups(Array.isArray(playgroundModels) ? playgroundModels : []), ], - providerModelGroups: [ + evaluationProviderModelGroups: [ (s) => [s.byokModels, s.providerKeys, s.failedByokProviderKeyIds], ( byokModels: ModelOption[], @@ -378,6 +418,34 @@ export const modelPickerLogic = kea([ }) }, ], + providerModelGroups: [ + (s) => [s.evaluationProviderModelGroups], + (groups: ProviderModelGroup[]): ProviderModelGroup[] => + groups.filter((group) => group.provider !== 'typesafe'), + ], + evaluationModelNotice: [ + (s) => [ + s.providerKeys, + s.providerKeysLoading, + s.byokModelsLoading, + s.failedByokProviderKeyIds, + s.evaluationProviderModelGroups, + ], + ( + providerKeys: LLMProviderKey[], + providerKeysLoading: boolean, + byokModelsLoading: boolean, + failedByokProviderKeyIds: string[], + groups: ProviderModelGroup[] + ): ByokModelNotice | null => + getByokModelNotice( + providerKeys, + providerKeysLoading, + byokModelsLoading, + failedByokProviderKeyIds, + groups + ), + ], byokModelNotice: [ (s) => [ s.providerKeys, @@ -393,17 +461,13 @@ export const modelPickerLogic = kea([ failedByokProviderKeyIds: string[], providerModelGroups: ProviderModelGroup[] ): ByokModelNotice | null => { - if (providerKeysLoading || byokModelsLoading) { - return null - } - const failedKeys = providerKeys.filter((key) => failedByokProviderKeyIds.includes(key.id)) - if (failedKeys.length > 0) { - return { kind: 'models-failed', keys: failedKeys } - } - if (providerModelGroups.some((group) => !group.disabledReason)) { - return null - } - return providerKeys.length === 0 ? { kind: 'no-keys' } : { kind: 'no-usable-keys' } + return getByokModelNotice( + providerKeys.filter((key) => key.provider !== 'typesafe'), + providerKeysLoading, + byokModelsLoading, + failedByokProviderKeyIds, + providerModelGroups + ) }, ], }), diff --git a/products/ai_observability/frontend/playground/llmPlaygroundModelLogic.ts b/products/ai_observability/frontend/playground/llmPlaygroundModelLogic.ts index 3453adc24a56..efca80bd18e4 100644 --- a/products/ai_observability/frontend/playground/llmPlaygroundModelLogic.ts +++ b/products/ai_observability/frontend/playground/llmPlaygroundModelLogic.ts @@ -204,7 +204,7 @@ export const llmPlaygroundModelLogic = kea([ ], modelPickerLogic, [ - 'byokModels', + 'generativeByokModels as byokModels', 'byokModelsLoading', 'playgroundModels', 'playgroundModelsLoading', @@ -422,7 +422,8 @@ export const llmPlaygroundModelLogic = kea([ loadProviderKeysFailure: () => resolvePendingTarget(), loadPlaygroundModelsFailure: () => resolvePendingTarget(), loadByokModelsFailure: () => resolvePendingTarget(), - loadByokModelsSuccess: ({ byokModels }: { byokModels: ModelOption[] }) => { + loadByokModelsSuccess: () => { + const byokModels = values.byokModels if (byokModels.length === 0) { return } diff --git a/products/ai_observability/frontend/settings/LLMProviderKeysSettings.tsx b/products/ai_observability/frontend/settings/LLMProviderKeysSettings.tsx index 972d6a5a7967..2d581f873f1e 100644 --- a/products/ai_observability/frontend/settings/LLMProviderKeysSettings.tsx +++ b/products/ai_observability/frontend/settings/LLMProviderKeysSettings.tsx @@ -102,6 +102,8 @@ function getKeyPlaceholder(provider: LLMProvider): string { return 'Enter your MiniMax API key' case 'zeabur': return 'sk-...' + case 'typesafe': + return 'Enter your TypeSafe API key' } } @@ -195,7 +197,7 @@ function AddKeyModal({ restrictionReason }: { restrictionReason: string | null } provider, name, api_key: apiKey, - set_as_active: !evaluationConfig?.active_provider_key, + set_as_active: provider !== 'typesafe' && !evaluationConfig?.active_provider_key, } if (isAzure) { payload.azure_endpoint = azureEndpoint @@ -229,7 +231,7 @@ function AddKeyModal({ restrictionReason }: { restrictionReason: string | null } provider, name, api_key: apiKey, - set_as_active: !evaluationConfig?.active_provider_key, + set_as_active: provider !== 'typesafe' && !evaluationConfig?.active_provider_key, } if (isAzure) { payload.azure_endpoint = azureEndpoint diff --git a/products/ai_observability/frontend/settings/llmProviderKeysLogic.ts b/products/ai_observability/frontend/settings/llmProviderKeysLogic.ts index 299adb970721..369dfdf1cb00 100644 --- a/products/ai_observability/frontend/settings/llmProviderKeysLogic.ts +++ b/products/ai_observability/frontend/settings/llmProviderKeysLogic.ts @@ -5,17 +5,10 @@ import api, { ApiError } from 'lib/api' import { lemonToast } from 'lib/lemon-ui/LemonToast/LemonToast' import { teamLogic } from 'scenes/teamLogic' +import type { LLMProviderEnumApi } from '../generated/api.schemas' + export type LLMProviderKeyState = 'unknown' | 'ok' | 'invalid' | 'error' -export type LLMProvider = - | 'openai' - | 'anthropic' - | 'gemini' - | 'openrouter' - | 'fireworks' - | 'azure_openai' - | 'together_ai' - | 'minimax' - | 'zeabur' +export type LLMProvider = LLMProviderEnumApi /** Default Azure OpenAI API version — keep in sync with backend DEFAULT_API_VERSION. */ export const DEFAULT_AZURE_API_VERSION = '2024-10-21' @@ -30,6 +23,7 @@ export const LLM_PROVIDER_LABELS: Record = { together_ai: 'Together AI', minimax: 'MiniMax', zeabur: 'Zeabur AI Hub', + typesafe: 'TypeSafe', } const LLM_PROVIDERS = new Set(Object.keys(LLM_PROVIDER_LABELS)) diff --git a/products/ai_observability/frontend/utils.test.ts b/products/ai_observability/frontend/utils.test.ts index 8b5a655c07e8..5a119d6472fc 100644 --- a/products/ai_observability/frontend/utils.test.ts +++ b/products/ai_observability/frontend/utils.test.ts @@ -75,6 +75,15 @@ function makeEvaluationRunRow({ } describe('mapEvaluationRunRow', () => { + it.each([0, 0.49, 1])('preserves a Jev probability of %s without inventing reasoning', (probability) => { + const row = makeEvaluationRunRow() + row[7] = '' + row[15] = probability + const run = mapEvaluationRunRow(row) + expect(run.probability).toBe(probability) + expect(run.reasoning).toBe('') + }) + it('maps sentiment rows without coercing missing boolean results to false', () => { const run = mapEvaluationRunRow( makeEvaluationRunRow({ diff --git a/products/ai_observability/frontend/utils.ts b/products/ai_observability/frontend/utils.ts index d0f3d50267e4..896c69cd9321 100644 --- a/products/ai_observability/frontend/utils.ts +++ b/products/ai_observability/frontend/utils.ts @@ -1127,6 +1127,7 @@ type RawEvaluationRunRow = [ sentiment_score: number | string | null, session_id: string | null, skipped: boolean | string | null, + probability?: number | string | null, ] export function normalizeEvaluationType(value: unknown): EvaluationType | undefined { @@ -1231,7 +1232,8 @@ export function mapEvaluationRunRow(row: RawEvaluationRunRow): EvaluationRun { session_id: row[13] || null, ...normalizedResult, skipped: isExplicitEvaluationPass(row[14]), - reasoning: row[7] || 'No reasoning provided', + reasoning: row[7] ?? '', + probability: normalizeOptionalNumber(row[15]), status: 'completed' as const, } } @@ -1275,7 +1277,8 @@ export async function queryEvaluationRuns(params: { properties.$ai_sentiment_label as sentiment_label, properties.$ai_sentiment_score as sentiment_score, properties.$ai_session_id as session_id, - properties.$ai_evaluation_skipped as skipped + properties.$ai_evaluation_skipped as skipped, + properties.$ai_evaluation_probability as probability FROM events WHERE event = '$ai_evaluation' diff --git a/services/mcp/src/api/generated.ts b/services/mcp/src/api/generated.ts index a2dc91309343..ec94dfa350c7 100644 --- a/services/mcp/src/api/generated.ts +++ b/services/mcp/src/api/generated.ts @@ -36051,6 +36051,7 @@ export namespace Schemas { * * `together_ai` - Together AI * * `minimax` - MiniMax * * `zeabur` - Zeabur AI Hub + * * `typesafe` - TypeSafe */ export type LLMProviderEnum = typeof LLMProviderEnum[keyof typeof LLMProviderEnum]; @@ -36065,6 +36066,7 @@ export namespace Schemas { TogetherAi: 'together_ai', Minimax: 'minimax', Zeabur: 'zeabur', + Typesafe: 'typesafe', } as const; /** @@ -51320,6 +51322,32 @@ export namespace Schemas { createdAt: string | null; } + /** + * * `openai` - Openai + * * `anthropic` - Anthropic + * * `gemini` - Gemini + * * `openrouter` - Openrouter + * * `fireworks` - Fireworks + * * `azure_openai` - Azure OpenAI + * * `together_ai` - Together AI + * * `minimax` - MiniMax + * * `zeabur` - Zeabur AI Hub + */ + export type LLMCompletionProviderEnum = typeof LLMCompletionProviderEnum[keyof typeof LLMCompletionProviderEnum]; + + + export const LLMCompletionProviderEnum = { + Openai: 'openai', + Anthropic: 'anthropic', + Gemini: 'gemini', + Openrouter: 'openrouter', + Fireworks: 'fireworks', + AzureOpenai: 'azure_openai', + TogetherAi: 'together_ai', + Minimax: 'minimax', + Zeabur: 'zeabur', + } as const; + export interface LLMModelInfo { /** Provider-specific model identifier (e.g. 'gpt-4o-mini', 'claude-3-5-sonnet-20241022'). */ id: string; @@ -63502,7 +63530,7 @@ export namespace Schemas { * * `together_ai` - Together AI * * `minimax` - MiniMax * * `zeabur` - Zeabur AI Hub */ - provider: LLMProviderEnum; + provider: LLMCompletionProviderEnum; /** * Provider model identifier to use for this tagger. * @maxLength 100 @@ -73486,7 +73514,7 @@ export namespace Schemas { * * `together_ai` - Together AI * * `minimax` - MiniMax * * `zeabur` - Zeabur AI Hub */ - provider: LLMProviderEnum; + provider: LLMCompletionProviderEnum; /** * Provider model identifier to use for this tagger. * @maxLength 100 @@ -106108,6 +106136,7 @@ export namespace Schemas { Openai: 'openai', Openrouter: 'openrouter', TogetherAi: 'together_ai', + Typesafe: 'typesafe', Zeabur: 'zeabur', } as const; diff --git a/services/mcp/src/generated/ai_observability/api.ts b/services/mcp/src/generated/ai_observability/api.ts index b9908e33273d..d3bae257e86b 100644 --- a/services/mcp/src/generated/ai_observability/api.ts +++ b/services/mcp/src/generated/ai_observability/api.ts @@ -740,9 +740,10 @@ export const EvaluationsCreateBody = () => zod 'together_ai', 'minimax', 'zeabur', + 'typesafe', ]) .describe( - '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub' + '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub\n\* `typesafe` - TypeSafe' ), model: zod.string().max(evaluationsCreateBodyModelConfigurationOneModelMax), provider_key_id: zod @@ -966,9 +967,10 @@ export const EvaluationsPartialUpdateBody = () => zod 'together_ai', 'minimax', 'zeabur', + 'typesafe', ]) .describe( - '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub' + '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub\n\* `typesafe` - TypeSafe' ), model: zod.string().max(evaluationsPartialUpdateBodyModelConfigurationOneModelMax), provider_key_id: zod @@ -1477,6 +1479,7 @@ export const LlmAnalyticsModelsRetrieveQueryParams = () => zod.object({ 'openai', 'openrouter', 'together_ai', + 'typesafe', 'zeabur', ]) .optional() From 98cf112a6cb78c80a352eadcd33ab05a97f84473 Mon Sep 17 00:00:00 2001 From: "tests-posthog[bot]" <250237707+tests-posthog[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 13:18:36 +0000 Subject: [PATCH 02/30] test(mcp): update unit test snapshots --- .../__snapshots__/tool-schemas/llma-evaluation-create.json | 5 +++-- .../tool-schemas/llma-evaluation-judge-models.json | 1 + .../__snapshots__/tool-schemas/llma-evaluation-update.json | 5 +++-- 3 files changed, 7 insertions(+), 4 deletions(-) diff --git a/services/mcp/tests/unit/__snapshots__/tool-schemas/llma-evaluation-create.json b/services/mcp/tests/unit/__snapshots__/tool-schemas/llma-evaluation-create.json index bbcc23cd980e..de7e987e8fce 100644 --- a/services/mcp/tests/unit/__snapshots__/tool-schemas/llma-evaluation-create.json +++ b/services/mcp/tests/unit/__snapshots__/tool-schemas/llma-evaluation-create.json @@ -112,7 +112,7 @@ "type": "string" }, "provider": { - "description": "* `openai` - Openai\n* `anthropic` - Anthropic\n* `gemini` - Gemini\n* `openrouter` - Openrouter\n* `fireworks` - Fireworks\n* `azure_openai` - Azure OpenAI\n* `together_ai` - Together AI\n* `minimax` - MiniMax\n* `zeabur` - Zeabur AI Hub", + "description": "* `openai` - Openai\n* `anthropic` - Anthropic\n* `gemini` - Gemini\n* `openrouter` - Openrouter\n* `fireworks` - Fireworks\n* `azure_openai` - Azure OpenAI\n* `together_ai` - Together AI\n* `minimax` - MiniMax\n* `zeabur` - Zeabur AI Hub\n* `typesafe` - TypeSafe", "enum": [ "openai", "anthropic", @@ -122,7 +122,8 @@ "azure_openai", "together_ai", "minimax", - "zeabur" + "zeabur", + "typesafe" ], "type": "string" }, diff --git a/services/mcp/tests/unit/__snapshots__/tool-schemas/llma-evaluation-judge-models.json b/services/mcp/tests/unit/__snapshots__/tool-schemas/llma-evaluation-judge-models.json index 51d148e109ab..cef5749759f5 100644 --- a/services/mcp/tests/unit/__snapshots__/tool-schemas/llma-evaluation-judge-models.json +++ b/services/mcp/tests/unit/__snapshots__/tool-schemas/llma-evaluation-judge-models.json @@ -16,6 +16,7 @@ "openai", "openrouter", "together_ai", + "typesafe", "zeabur" ], "type": "string" diff --git a/services/mcp/tests/unit/__snapshots__/tool-schemas/llma-evaluation-update.json b/services/mcp/tests/unit/__snapshots__/tool-schemas/llma-evaluation-update.json index e87591b624f1..e2f6c7616058 100644 --- a/services/mcp/tests/unit/__snapshots__/tool-schemas/llma-evaluation-update.json +++ b/services/mcp/tests/unit/__snapshots__/tool-schemas/llma-evaluation-update.json @@ -115,7 +115,7 @@ "type": "string" }, "provider": { - "description": "* `openai` - Openai\n* `anthropic` - Anthropic\n* `gemini` - Gemini\n* `openrouter` - Openrouter\n* `fireworks` - Fireworks\n* `azure_openai` - Azure OpenAI\n* `together_ai` - Together AI\n* `minimax` - MiniMax\n* `zeabur` - Zeabur AI Hub", + "description": "* `openai` - Openai\n* `anthropic` - Anthropic\n* `gemini` - Gemini\n* `openrouter` - Openrouter\n* `fireworks` - Fireworks\n* `azure_openai` - Azure OpenAI\n* `together_ai` - Together AI\n* `minimax` - MiniMax\n* `zeabur` - Zeabur AI Hub\n* `typesafe` - TypeSafe", "enum": [ "openai", "anthropic", @@ -125,7 +125,8 @@ "azure_openai", "together_ai", "minimax", - "zeabur" + "zeabur", + "typesafe" ], "type": "string" }, From 6703c4dee688da883acfbfb96607f697b409676c Mon Sep 17 00:00:00 2001 From: Bernat Torres Date: Fri, 25 Sep 2026 12:46:50 +0200 Subject: [PATCH 03/30] feat(aio): support compatible System One judge endpoints --- .../internal/ai-observability-judge-inputs.md | 20 +- .../ai_observability/evaluation_llm_judge.py | 21 +- .../ai_observability/test_run_evaluation.py | 32 ++- .../backend/api/evaluations.py | 7 +- .../backend/api/provider_keys.py | 82 +++++++- .../ai_observability/backend/api/proxy.py | 10 +- .../backend/api/test/test_evaluations.py | 6 +- .../backend/api/test/test_provider_keys.py | 65 +++++- .../ai_observability/backend/llm/client.py | 8 +- .../backend/llm/system_one.py | 158 ++++++++++++++ .../backend/llm/test/test_system_one.py | 193 ++++++++++++++++++ .../backend/llm/test/test_typesafe.py | 122 ----------- .../ai_observability/backend/llm/typesafe.py | 120 ----------- .../backend/models/model_configuration.py | 2 +- .../backend/models/provider_keys.py | 6 +- .../frontend/generated/api.schemas.ts | 36 +++- .../frontend/generated/api.zod.ts | 46 ++++- .../frontend/modelPickerLogic.test.ts | 67 +++--- .../settings/LLMProviderKeysSettings.tsx | 81 +++++++- .../SystemOneConnectionFields.stories.tsx | 28 +++ .../settings/SystemOneConnectionFields.tsx | 48 +++++ .../settings/llmProviderKeysLogic.test.ts | 2 + .../frontend/settings/llmProviderKeysLogic.ts | 43 +++- services/mcp/src/api/generated.ts | 36 +++- .../mcp/src/generated/ai_observability/api.ts | 4 +- 25 files changed, 904 insertions(+), 339 deletions(-) create mode 100644 products/ai_observability/backend/llm/system_one.py create mode 100644 products/ai_observability/backend/llm/test/test_system_one.py delete mode 100644 products/ai_observability/backend/llm/test/test_typesafe.py delete mode 100644 products/ai_observability/backend/llm/typesafe.py create mode 100644 products/ai_observability/frontend/settings/SystemOneConnectionFields.stories.tsx create mode 100644 products/ai_observability/frontend/settings/SystemOneConnectionFields.tsx diff --git a/docs/internal/ai-observability-judge-inputs.md b/docs/internal/ai-observability-judge-inputs.md index a8682d2eac3e..ef8c8338e7f4 100644 --- a/docs/internal/ai-observability-judge-inputs.md +++ b/docs/internal/ai-observability-judge-inputs.md @@ -34,12 +34,22 @@ They sample the combined input, tool definitions, and output only when that text Implementation: [trace judge](../../posthog/temporal/ai_observability/run_trace_evaluation.py), [session judge](../../posthog/temporal/ai_observability/run_session_evaluation.py), and [generation judge](../../posthog/temporal/ai_observability/evaluation_llm_judge.py). -## Jev boolean judge +## System One boolean judges -Jev is available under the existing LLM judge option with a customer-provided TypeSafe API key. -Select the key and `jev-1.13.0` on each evaluation; TypeSafe keys cannot become the shared active provider key used by other AI features. +System One-compatible models are available under the existing LLM judge option. +Add a connection under **System One (Jev)** in provider key settings. +The default endpoint is TypeSafe's `https://api.typesafe.ai/v1`, with model `jev-1.13.0` and a TypeSafe API key. +Advanced configuration accepts a different public HTTPS base URL and model ID for compatible services. +The client appends `/systemone` to the base URL and sends the API key as a bearer token. +An empty key selects no authentication for a custom endpoint; TypeSafe requires a key. +Changing the endpoint requires entering its credential again, or explicitly choosing no authentication, so an existing key is not forwarded to a new host. +Private network destinations and redirects are blocked by the shared DNS-pinned HTTP transport. +Saving a connection validates it with a short synthetic input and two Noul questions, including applicability, without sending evaluation data. +Select the connection and configured model on each evaluation; these connections cannot become the shared active provider key used by other AI features. Provider keys keep the provider they were created with; switching providers requires a new key. -Jev supports boolean evaluations only and uses the same formatted text for generation, trace, and session targets. +This integration supports boolean evaluations only and uses the same formatted text for generation, trace, and session targets. +API compatibility does not guarantee equivalent judgments or calibration across models. +Compare results on representative inputs when changing models. The evaluation prompt becomes a [Noul question](https://docs.typesafe.ai/primitives/noul). A probability of at least 0.5 produces `true`; the evaluation's existing pass/fail polarity still applies. @@ -48,7 +58,7 @@ Uncertainty alone does not produce N/A. The raw probability is stored in `$ai_evaluation_probability`, with token usage and the resolved model version. Jev provides no written reasoning, so reports inspect the original source when explaining outcomes. -TypeSafe rate limits and overload responses are retried through Temporal, honoring `Retry-After` up to five minutes. +Rate limits and overload responses are retried through Temporal, honoring `Retry-After` up to five minutes. If retries fail, the run fails and the evaluation stays enabled. Invalid probabilities or missing answers fail the evaluation rather than producing a false result. Inputs rejected for exceeding the model's context window are skipped. diff --git a/posthog/temporal/ai_observability/evaluation_llm_judge.py b/posthog/temporal/ai_observability/evaluation_llm_judge.py index 76386be60251..e95a5da3b76d 100644 --- a/posthog/temporal/ai_observability/evaluation_llm_judge.py +++ b/posthog/temporal/ai_observability/evaluation_llm_judge.py @@ -46,8 +46,8 @@ RateLimitError, StructuredOutputParseError, ) +from products.ai_observability.backend.llm.system_one import SystemOneClient, SystemOneRateLimitError from products.ai_observability.backend.llm.types import CompletionResponse, Usage -from products.ai_observability.backend.llm.typesafe import TypeSafeClient, TypeSafeRateLimitError from products.ai_observability.backend.text_repr.formatters import add_line_numbers, reduce_by_uniform_sampling logger = structlog.get_logger(__name__) @@ -343,15 +343,18 @@ def call_llm_judge( if provider == "typesafe": if provider_key is not None and provider_key.provider != provider: raise ProviderMismatchError(provider_key.provider, provider) - typesafe_result = TypeSafeClient.evaluate_boolean( + system_one_result = SystemOneClient.evaluate_boolean( api_key=provider_key.encrypted_config.get("api_key", "") if provider_key else "", + base_url=provider_key.encrypted_config.get("base_url", SystemOneClient.BASE_URL) + if provider_key + else SystemOneClient.BASE_URL, model=model, prompt=evaluation["evaluation_config"]["prompt"], source=user_prompt, allows_na=allows_na, ) - probability = typesafe_result.answers["verdict"].noul - applicable = not allows_na or typesafe_result.answers["applicable"].noul >= 0.5 + probability = system_one_result.answers["verdict"].noul + applicable = not allows_na or system_one_result.answers["applicable"].noul >= 0.5 parsed: BooleanEvalResult | BooleanWithNAEvalResult = ( BooleanWithNAEvalResult( reasoning="", applicable=applicable, verdict=probability >= 0.5 if applicable else None @@ -359,15 +362,15 @@ def call_llm_judge( if allows_na else BooleanEvalResult(reasoning="", verdict=probability >= 0.5) ) - model = typesafe_result.model + model = system_one_result.model response = CompletionResponse( content="", model=model, parsed=parsed, usage=Usage( - input_tokens=typesafe_result.usage.input_tokens, - output_tokens=typesafe_result.usage.output_tokens, - total_tokens=typesafe_result.usage.input_tokens + typesafe_result.usage.output_tokens, + input_tokens=system_one_result.usage.input_tokens, + output_tokens=system_one_result.usage.output_tokens, + total_tokens=system_one_result.usage.input_tokens + system_one_result.usage.output_tokens, ), ) else: @@ -380,7 +383,7 @@ def call_llm_judge( response_format=response_format, ) ) - except TypeSafeRateLimitError as e: + except SystemOneRateLimitError as e: increment_errors("rate_limit", provider=provider) raise ApplicationError( str(e), diff --git a/posthog/temporal/ai_observability/test_run_evaluation.py b/posthog/temporal/ai_observability/test_run_evaluation.py index 601c77e6ae70..d1dc346c3d5d 100644 --- a/posthog/temporal/ai_observability/test_run_evaluation.py +++ b/posthog/temporal/ai_observability/test_run_evaluation.py @@ -84,15 +84,32 @@ def _mock_config_with_active_key(provider: str = "openai") -> MagicMock: return MagicMock(active_provider_key=key) +@pytest.mark.parametrize( + "connection_config,base_url,model", + [ + ({"api_key": "test-typesafe-key"}, "https://api.typesafe.ai/v1", "jev-1.13.0"), + ( + {"api_key": "", "base_url": "https://decisions.example.com/v1"}, + "https://decisions.example.com/v1", + "custom-model", + ), + ], +) @pytest.mark.parametrize( "probability,applicability,allows_na,verdict", [(0.49, 1.0, False, False), (0.5, 1.0, False, True), (0.9, 0.1, True, None), (0.0, 0.9, True, False)], ) def test_typesafe_judge_emits_boolean_probability_without_reasoning( - probability: float, applicability: float, allows_na: bool, verdict: bool | None + probability: float, + applicability: float, + allows_na: bool, + verdict: bool | None, + connection_config: dict[str, str], + base_url: str, + model: str, ) -> None: - key = MagicMock(provider="typesafe", encrypted_config={"api_key": "test-typesafe-key"}) - resolved = MagicMock(provider="typesafe", model="jev-1.13.0", provider_key=key, is_byok=True) + key = MagicMock(provider="typesafe", encrypted_config=connection_config) + resolved = MagicMock(provider="typesafe", model=model, provider_key=key, is_byok=True) response = MagicMock(status_code=200) response.json.return_value = { "model": "jev-1.13.0", @@ -110,7 +127,7 @@ def test_typesafe_judge_emits_boolean_probability_without_reasoning( } with ( patch("posthog.temporal.ai_observability.evaluation_llm_judge.model_spec") as spec, - patch("requests.request", return_value=response), + patch("products.ai_observability.backend.llm.system_one.pinned_request", return_value=response) as request, ): spec.return_value.resolve.return_value = resolved result = call_llm_judge( @@ -120,6 +137,8 @@ def test_typesafe_judge_emits_boolean_probability_without_reasoning( allows_na=allows_na, ) + assert request.call_args.args[1] == f"{base_url}/systemone" + assert request.call_args.kwargs["json"]["model"] == model assert result["verdict"] is verdict assert result["reasoning"] == "" assert result["probability"] == probability @@ -136,7 +155,10 @@ def test_typesafe_rate_limit_retries_without_disabling_the_evaluation() -> None: key = MagicMock(provider="typesafe", encrypted_config={"api_key": "test-typesafe-key"}) with ( patch("posthog.temporal.ai_observability.evaluation_llm_judge.model_spec") as spec, - patch("requests.request", return_value=MagicMock(status_code=429, headers={"Retry-After": "15"})), + patch( + "products.ai_observability.backend.llm.system_one.pinned_request", + return_value=MagicMock(status_code=429, headers={"Retry-After": "15"}), + ), pytest.raises(ApplicationError) as error, ): spec.return_value.resolve.return_value = MagicMock( diff --git a/products/ai_observability/backend/api/evaluations.py b/products/ai_observability/backend/api/evaluations.py index ccc5c5b84856..5471a8548bc7 100644 --- a/products/ai_observability/backend/api/evaluations.py +++ b/products/ai_observability/backend/api/evaluations.py @@ -43,7 +43,6 @@ from ..evaluation_conditions import build_condition_filter from ..hog import compile_ai_observability_hog from ..llm import DEFAULT_MODEL_BY_PROVIDER -from ..llm.typesafe import TypeSafeClient from ..models.evaluation_config import EvaluationConfig from ..models.evaluation_configs import ( EVALUATION_TEST_LOOKBACK_DAYS, @@ -253,10 +252,10 @@ def validate(self, data: dict[str, Any]) -> dict[str, Any]: if errors: raise serializers.ValidationError(errors, code="required") if data["provider"] == LLMProvider.TYPESAFE: - if data["model"] != TypeSafeClient.MODEL: - raise serializers.ValidationError({"model": "Select a supported Jev model."}) if not data.get("provider_key_id"): - raise serializers.ValidationError({"provider_key_id": "Select a TypeSafe API key for this evaluation."}) + raise serializers.ValidationError( + {"provider_key_id": "Select a System One connection for this evaluation."} + ) return data def get_provider_key_name(self, obj: LLMModelConfiguration) -> str | None: diff --git a/products/ai_observability/backend/api/provider_keys.py b/products/ai_observability/backend/api/provider_keys.py index c8637d3608af..2289420f4ea4 100644 --- a/products/ai_observability/backend/api/provider_keys.py +++ b/products/ai_observability/backend/api/provider_keys.py @@ -30,6 +30,7 @@ error_field_for_validation_message, is_allowed_azure_endpoint, ) +from ..llm.system_one import SystemOneClient from ..models.evaluation_config import EvaluationConfig from ..models.evaluations import Evaluation from ..models.model_configuration import LLMModelConfiguration @@ -90,7 +91,18 @@ def _validation_error_field(provider: str, error_message: str | None) -> str: class LLMProviderKeySerializer(serializers.ModelSerializer): - api_key = serializers.CharField(write_only=True, required=False) + api_key = serializers.CharField(write_only=True, required=False, allow_blank=True) + base_url = serializers.URLField( + write_only=True, required=False, help_text="System One API base URL, including /v1. Defaults to TypeSafe." + ) + system_one_model = serializers.CharField( + write_only=True, + required=False, + max_length=100, + help_text="Model ID served by the System One endpoint. Defaults to jev-1.13.0.", + ) + base_url_display = serializers.SerializerMethodField(help_text="Configured System One base URL.") + system_one_model_display = serializers.SerializerMethodField(help_text="Configured System One model ID.") api_key_masked = serializers.SerializerMethodField() azure_endpoint = serializers.URLField(write_only=True, required=False, help_text="Azure OpenAI endpoint URL") api_version = serializers.CharField( @@ -111,6 +123,10 @@ class Meta: "error_message", "api_key", "api_key_masked", + "base_url", + "system_one_model", + "base_url_display", + "system_one_model_display", "azure_endpoint", "api_version", "azure_endpoint_display", @@ -125,6 +141,24 @@ class Meta: def get_api_key_masked(self, obj: LLMProviderKey) -> str: return mask_key_value(obj.encrypted_config.get("api_key", "")) + def get_base_url_display(self, obj: LLMProviderKey) -> str | None: + return ( + obj.encrypted_config.get("base_url", SystemOneClient.BASE_URL) + if obj.provider == LLMProvider.TYPESAFE + else None + ) + + def get_system_one_model_display(self, obj: LLMProviderKey) -> str | None: + return ( + obj.encrypted_config.get("model", SystemOneClient.MODEL) if obj.provider == LLMProvider.TYPESAFE else None + ) + + def validate_base_url(self, value: str) -> str: + try: + return SystemOneClient.normalize_base_url(value) + except ValueError as error: + raise serializers.ValidationError(str(error)) from error + def get_azure_endpoint_display(self, obj: LLMProviderKey) -> str | None: if obj.provider != LLMProvider.AZURE_OPENAI: return None @@ -159,6 +193,19 @@ def validate(self, data): raise serializers.ValidationError({"api_key": "API key is required when creating a new provider key."}) provider = data.get("provider", getattr(self.instance, "provider", None)) + if provider != LLMProvider.TYPESAFE: + if "base_url" in data or "system_one_model" in data: + raise serializers.ValidationError( + {"base_url": "These settings are only available for System One connections."} + ) + if data.get("api_key") == "": + raise serializers.ValidationError({"api_key": "An API key is required."}) + elif self.instance is not None and "base_url" in data: + current_url = self.instance.encrypted_config.get("base_url", SystemOneClient.BASE_URL) + if data["base_url"] != current_url and "api_key" not in data: + raise serializers.ValidationError( + {"api_key": "Enter the credential for the new endpoint, or an empty value for no authentication."} + ) if self.instance is not None and provider != self.instance.provider: raise serializers.ValidationError({"provider": "A key's provider cannot change. Create a new key instead."}) if provider == LLMProvider.TYPESAFE and data.get("set_as_active"): @@ -176,6 +223,12 @@ def validate(self, data): return data + def _system_one_config(self, validated_data: dict, current: dict | None = None) -> dict: + config = dict(current or {}) + config["base_url"] = validated_data.pop("base_url", config.get("base_url", SystemOneClient.BASE_URL)) + config["model"] = validated_data.pop("system_one_model", config.get("model", SystemOneClient.MODEL)) + return config + def _pop_azure_kwargs(self, validated_data: dict) -> dict: """Pop Azure-specific write-only fields out of ``validated_data`` and return them as kwargs. @@ -217,6 +270,16 @@ def create(self, validated_data): azure_kwargs = self._normalize_azure_config(provider, azure_kwargs) + if provider == LLMProvider.TYPESAFE: + connection_config = self._system_one_config(validated_data) + state, error_message = validate_provider_key(provider, api_key or "", **connection_config) + if state != LLMProviderKey.State.OK: + raise serializers.ValidationError({"api_key": error_message}) + validated_data["encrypted_config"] = {"api_key": api_key or "", **connection_config} + validated_data["state"] = state + validated_data["error_message"] = None + return super().create(validated_data) + if api_key: state, error_message = validate_provider_key(provider, api_key, **azure_kwargs) if state != LLMProviderKey.State.OK: @@ -236,6 +299,17 @@ def create(self, validated_data): return instance def update(self, instance, validated_data): + if instance.provider == LLMProvider.TYPESAFE: + if any(field in validated_data for field in ("api_key", "base_url", "system_one_model")): + config = self._system_one_config(validated_data, instance.encrypted_config) + config["api_key"] = validated_data.pop("api_key", config.get("api_key", "")) + state, error_message = validate_provider_key(instance.provider, **config) + if state != LLMProviderKey.State.OK: + raise serializers.ValidationError({"api_key": error_message}) + instance.encrypted_config = config + instance.state = state + instance.error_message = None + return super().update(instance, validated_data) api_key = validated_data.pop("api_key", None) azure_kwargs = self._normalize_azure_config(instance.provider, self._pop_azure_kwargs(validated_data)) @@ -358,13 +432,15 @@ def validate(self, request: Request, **_kwargs) -> Response: instance = self.get_object() api_key = instance.encrypted_config.get("api_key") - if not api_key: + if not api_key and instance.provider != LLMProvider.TYPESAFE: return Response( {"detail": "No API key configured for this provider key."}, status=status.HTTP_400_BAD_REQUEST, ) - state, error_message = validate_provider_key(instance.provider, api_key, **instance.provider_extra_kwargs()) + state, error_message = validate_provider_key( + instance.provider, api_key or "", **instance.provider_extra_kwargs() + ) instance.state = state instance.error_message = error_message instance.save(update_fields=["state", "error_message"]) diff --git a/products/ai_observability/backend/api/proxy.py b/products/ai_observability/backend/api/proxy.py index f1f0d27dc1c6..77bee39e080c 100644 --- a/products/ai_observability/backend/api/proxy.py +++ b/products/ai_observability/backend/api/proxy.py @@ -49,7 +49,11 @@ get_playground_models, ) from products.ai_observability.backend.llm.errors import UnsupportedProviderError -from products.ai_observability.backend.models.provider_keys import LLMProviderKey, llm_completion_provider_choices +from products.ai_observability.backend.models.provider_keys import ( + LLMProvider, + LLMProviderKey, + llm_completion_provider_choices, +) from ee.hogai.utils.asgi import SyncIterableToAsync @@ -63,7 +67,7 @@ def models_cache_key(provider_key_id: str | uuid.UUID) -> str: PROVIDER_DISPLAY_NAMES: dict[str, str] = { - "typesafe": "TypeSafe", + "typesafe": "System One (Jev)", "openai": "OpenAI", "anthropic": "Anthropic", "gemini": "Gemini", @@ -190,7 +194,7 @@ def _get_provider_key( raise ValueError("Provider key not found") api_key = key.encrypted_config.get("api_key") - if not api_key: + if not api_key and key.provider != LLMProvider.TYPESAFE: raise ValueError("No API key configured for this provider key") if touch_last_used: diff --git a/products/ai_observability/backend/api/test/test_evaluations.py b/products/ai_observability/backend/api/test/test_evaluations.py index 004d55bc3ea4..64a6eaf10c64 100644 --- a/products/ai_observability/backend/api/test/test_evaluations.py +++ b/products/ai_observability/backend/api/test/test_evaluations.py @@ -57,13 +57,11 @@ class TestModelConfigurationSerializer(SimpleTestCase): @parameterized.expand( [ ("missing_key", "jev-1.13.0", None, False), - ("unsupported_model", "other-model", str(uuid4()), False), + ("custom_model", "other-model", str(uuid4()), True), ("configured", "jev-1.13.0", str(uuid4()), True), ] ) - def test_typesafe_requires_supported_model_and_explicit_key( - self, _name: str, model: str, key_id: str | None, valid: bool - ) -> None: + def test_system_one_requires_explicit_key(self, _name: str, model: str, key_id: str | None, valid: bool) -> None: serializer = ModelConfigurationSerializer( data={"provider": "typesafe", "model": model, "provider_key_id": key_id} ) diff --git a/products/ai_observability/backend/api/test/test_provider_keys.py b/products/ai_observability/backend/api/test/test_provider_keys.py index f840c2af3209..3ef5deaa9d5f 100644 --- a/products/ai_observability/backend/api/test/test_provider_keys.py +++ b/products/ai_observability/backend/api/test/test_provider_keys.py @@ -17,7 +17,9 @@ from products.ai_observability.backend.api.provider_keys import LLMProviderKeySerializer from products.ai_observability.backend.api.proxy import LLMProxyCompletionSerializer, models_cache_key from products.ai_observability.backend.api.taggers import TaggerModelConfigurationWriteSerializer +from products.ai_observability.backend.llm.client import Client from products.ai_observability.backend.llm.providers.azure_openai import DEFAULT_API_VERSION +from products.ai_observability.backend.llm.system_one import SystemOneClient from products.ai_observability.backend.models.evaluation_config import EvaluationConfig from products.ai_observability.backend.models.evaluations import Evaluation from products.ai_observability.backend.models.model_configuration import LLMModelConfiguration @@ -54,6 +56,12 @@ def test_typesafe_cannot_be_created_as_the_shared_key(self) -> None: self.assertFalse(serializer.is_valid()) self.assertIn("set_as_active", serializer.errors) + @parameterized.expand([("https://decisions.example.com/v1", False), (SystemOneClient.BASE_URL, True)]) + def test_endpoint_change_requires_explicit_credentials(self, base_url: str, valid: bool) -> None: + key = LLMProviderKey(provider="typesafe", encrypted_config={"api_key": "example-token"}) + serializer = LLMProviderKeySerializer(key, data={"base_url": base_url}, partial=True) + self.assertEqual(serializer.is_valid(), valid, serializer.errors) + def _setup_team(): org = Organization.objects.create(name="test") @@ -140,7 +148,62 @@ def test_can_create_provider_key(self, provider: str, mock_validate: Mock) -> No self.assertEqual(response.data["api_key_masked"], "sk-t...2345") self.assertNotIn("api_key", response.data) - mock_validate.assert_called_once_with(provider, "sk-test-key-12345") + expected_config = ( + {"base_url": SystemOneClient.BASE_URL, "model": SystemOneClient.MODEL} if provider == "typesafe" else {} + ) + mock_validate.assert_called_once_with(provider, "sk-test-key-12345", **expected_config) + + @patch("products.ai_observability.backend.llm.system_one.pinned_request") + def test_custom_system_one_connection_round_trip(self, request: Mock) -> None: + request.return_value = Mock(status_code=200) + request.return_value.json.return_value = { + "model": "custom-model", + "answers": {"verdict": {"type": "noul", "noul": 1.0}, "applicable": {"type": "noul", "noul": 1.0}}, + "usage": {"input_tokens": 12, "output_tokens": 0}, + } + url = f"/api/environments/{self.team.id}/llm_analytics/provider_keys/" + response = self.client.post( + url, + { + "provider": "typesafe", + "name": "Custom", + "api_key": "", + "base_url": "https://decisions.example.com/v1/", + "system_one_model": "custom-model", + }, + ) + self.assertEqual(response.status_code, 201, response.data) + self.assertEqual(response.data["base_url_display"], "https://decisions.example.com/v1") + key = LLMProviderKey.objects.get(id=response.data["id"]) + self.assertEqual(Client.list_models("typesafe", **key.provider_extra_kwargs()), ["custom-model"]) + model_config = LLMModelConfiguration(provider="typesafe", model="custom-model", provider_key=key) + self.assertEqual(model_config.get_available_models(), ["custom-model"]) + self.assertEqual(request.call_args.kwargs["headers"], {}) + request.reset_mock() + response = self.client.patch(f"{url}{key.id}/", {"base_url": "https://other.example.com/v1"}) + self.assertEqual(response.status_code, 400) + request.assert_not_called() + key.refresh_from_db() + self.assertEqual(key.encrypted_config["base_url"], "https://decisions.example.com/v1") + request.return_value.status_code = 401 + response = self.client.patch( + f"{url}{key.id}/", {"base_url": "https://other.example.com/v1", "api_key": "fake-token"} + ) + self.assertEqual(response.status_code, 400) + key.refresh_from_db() + self.assertEqual(key.encrypted_config["base_url"], "https://decisions.example.com/v1") + self.assertEqual(key.encrypted_config["api_key"], "") + request.return_value.status_code = 200 + response = self.client.patch( + f"{url}{key.id}/", {"base_url": "https://other.example.com/v1", "api_key": "fake-token"} + ) + self.assertEqual(response.status_code, 200, response.data) + key.refresh_from_db() + self.assertEqual(key.encrypted_config["base_url"], "https://other.example.com/v1") + self.assertEqual(request.call_args.kwargs["headers"], {"Authorization": "Bearer fake-token"}) + response = self.client.post(f"{url}{key.id}/validate/") + self.assertEqual(response.status_code, 200, response.data) + self.assertEqual(request.call_args.args[1], "https://other.example.com/v1/systemone") @patch("products.ai_observability.backend.api.provider_keys.validate_provider_key") def test_can_create_provider_key_with_set_as_active(self, mock_validate): diff --git a/products/ai_observability/backend/llm/client.py b/products/ai_observability/backend/llm/client.py index ad85f43a247a..6227f80a2b01 100644 --- a/products/ai_observability/backend/llm/client.py +++ b/products/ai_observability/backend/llm/client.py @@ -9,13 +9,13 @@ from typing import TYPE_CHECKING, Any from products.ai_observability.backend.llm.errors import ProviderMismatchError, UnsupportedProviderError +from products.ai_observability.backend.llm.system_one import SystemOneClient from products.ai_observability.backend.llm.types import ( AnalyticsContext, CompletionRequest, CompletionResponse, StreamChunk, ) -from products.ai_observability.backend.llm.typesafe import TypeSafeClient if TYPE_CHECKING: from products.ai_observability.backend.llm.config import ProviderConfig @@ -83,21 +83,21 @@ def _resolve_credentials(self) -> tuple[str | None, str | None]: def validate_key(cls, provider: str, api_key: str, **kwargs: Any) -> tuple[str, str | None]: """Validate an API key for a provider. Returns (state, error_message).""" if provider == "typesafe": - return TypeSafeClient.validate_key(api_key) + return SystemOneClient.validate_key(api_key, **kwargs) return _get_provider(provider).validate_key(api_key, **kwargs) @classmethod def list_models(cls, provider: str, api_key: str | None = None, **kwargs: Any) -> list[str]: """List available models for a provider.""" if provider == "typesafe": - return [TypeSafeClient.MODEL] + return [kwargs.get("model", SystemOneClient.MODEL)] return _get_provider(provider).list_models(api_key, **kwargs) @classmethod def recommended_models(cls, provider: str) -> set[str]: """Return the set of curated/recommended model IDs for a provider.""" if provider == "typesafe": - return {TypeSafeClient.MODEL} + return {SystemOneClient.MODEL} return _get_provider(provider).recommended_models() diff --git a/products/ai_observability/backend/llm/system_one.py b/products/ai_observability/backend/llm/system_one.py new file mode 100644 index 000000000000..76b74da2dca8 --- /dev/null +++ b/products/ai_observability/backend/llm/system_one.py @@ -0,0 +1,158 @@ +import math +from datetime import UTC, datetime +from email.utils import parsedate_to_datetime +from typing import Literal +from urllib.parse import urlsplit + +import requests +from pydantic import BaseModel, Field, ValidationError + +from posthog.security.pinned_requests import SSRFBlockedError, pinned_request +from posthog.security.url_validation import has_authority_bypass_chars + +from products.ai_observability.backend.llm.errors import ( + AuthenticationError, + ContextWindowExceededError, + ModelNotFoundError, + ModelPermissionError, + ProviderConnectionError, + RateLimitError, + StructuredOutputParseError, + is_context_window_error_message, +) + + +class NoulAnswer(BaseModel): + type: Literal["noul"] + noul: float = Field(strict=True, ge=0, le=1, allow_inf_nan=False) + + +class SystemOneUsage(BaseModel): + input_tokens: int = Field(strict=True, ge=0) + output_tokens: int = Field(strict=True, ge=0) + + +class SystemOneResponse(BaseModel): + model: str = Field(min_length=1) + answers: dict[str, NoulAnswer] + usage: SystemOneUsage + + +class SystemOneRateLimitError(RateLimitError): + def __init__(self, retry_after: str | None) -> None: + super().__init__("The System One endpoint is busy. Try again later.") + self.retry_after: float | None = None + if retry_after: + try: + delay = float(retry_after) + except ValueError: + try: + delay = (parsedate_to_datetime(retry_after) - datetime.now(UTC)).total_seconds() + except (ValueError, TypeError, OverflowError): + return + if math.isfinite(delay): + self.retry_after = max(1, min(delay, 300)) + + +class SystemOneClient: + MODEL = "jev-1.13.0" + BASE_URL = "https://api.typesafe.ai/v1" + + @staticmethod + def normalize_base_url(base_url: str) -> str: + parsed = urlsplit(base_url) + if ( + has_authority_bypass_chars(base_url) + or parsed.scheme != "https" + or not parsed.hostname + or parsed.username is not None + or parsed.password is not None + or parsed.query + or parsed.fragment + ): + raise ValueError("Use an HTTPS base URL without credentials, a query, or a fragment.") + return base_url.rstrip("/") + + @staticmethod + def evaluate_boolean( + *, api_key: str, model: str, prompt: str, source: str, allows_na: bool, base_url: str = BASE_URL + ) -> SystemOneResponse: + base_url = SystemOneClient.normalize_base_url(base_url) + if not api_key and base_url == SystemOneClient.BASE_URL: + raise AuthenticationError("A TypeSafe API key is required.") + questions = {"verdict": {"type": "noul", "instructions": prompt}} + if allows_na: + questions["applicable"] = { + "type": "noul", + "instructions": ( + "Do these evaluation criteria apply to this input? Answer true when the criteria can be " + "evaluated, even if they are not met. Answer false only when they are not relevant.\n\n" + prompt + ), + } + try: + response = pinned_request( + "POST", + f"{base_url}/systemone", + headers={"Authorization": f"Bearer {api_key}"} if api_key else {}, + json={"model": model, "state": source, "questions": questions}, + timeout=60, + ) + except SSRFBlockedError as error: + raise StructuredOutputParseError("This endpoint is not allowed. Use a public HTTPS endpoint.") from error + except requests.RequestException as error: + raise ProviderConnectionError("Could not reach the System One endpoint. Try again.") from error + + if response.status_code == 401: + raise AuthenticationError("The endpoint rejected this credential. Check the bearer token.") + if response.status_code == 403: + raise ModelPermissionError(model) + if response.status_code == 404: + raise ModelNotFoundError(model) + if response.status_code in (429, 503, 529): + raise SystemOneRateLimitError(response.headers.get("Retry-After")) + if response.status_code >= 500: + raise ProviderConnectionError("The System One endpoint is temporarily unavailable. Try again.") + if response.status_code == 413 or ( + response.status_code == 422 and is_context_window_error_message(response.text) + ): + raise ContextWindowExceededError("This input exceeds the endpoint's size limit. Reduce the input.") + if response.status_code != 200: + raise StructuredOutputParseError( + "The endpoint rejected the evaluation request. Check the model and criteria." + ) + try: + result = SystemOneResponse.model_validate(response.json()) + except (ValidationError, ValueError) as error: + raise StructuredOutputParseError( + "The endpoint returned an invalid System One response. Check compatibility." + ) from error + # boffin: Missing answers cannot become a false verdict. + if not questions.keys() <= result.answers.keys(): + raise StructuredOutputParseError( + "The endpoint did not answer every evaluation question. Check compatibility." + ) + return result + + @staticmethod + def validate_key(api_key: str, *, base_url: str = BASE_URL, model: str = MODEL) -> tuple[str, str | None]: + try: + SystemOneClient.evaluate_boolean( + api_key=api_key, + base_url=base_url, + model=model, + prompt="Does the text contain a greeting?", + source="Hello!", + allows_na=True, + ) + except (AuthenticationError, ModelPermissionError) as error: + return "invalid", str(error) + except ( + ValueError, + ModelNotFoundError, + ProviderConnectionError, + RateLimitError, + StructuredOutputParseError, + ContextWindowExceededError, + ) as error: + return "error", str(error) + return "ok", None diff --git a/products/ai_observability/backend/llm/test/test_system_one.py b/products/ai_observability/backend/llm/test/test_system_one.py new file mode 100644 index 000000000000..1f56f2cad602 --- /dev/null +++ b/products/ai_observability/backend/llm/test/test_system_one.py @@ -0,0 +1,193 @@ +import pytest +from unittest.mock import Mock, patch + +from django.test import override_settings + +from products.ai_observability.backend.llm.client import Client +from products.ai_observability.backend.llm.errors import ( + AuthenticationError, + ContextWindowExceededError, + ModelPermissionError, + ProviderConnectionError, + StructuredOutputParseError, +) +from products.ai_observability.backend.llm.system_one import SystemOneClient, SystemOneRateLimitError + + +@pytest.mark.parametrize("status, expected_state", [(200, "ok"), (401, "invalid"), (403, "invalid"), (500, "error")]) +def test_typesafe_key_validation(status: int, expected_state: str) -> None: + response = Mock(status_code=status) + response.json.return_value = { + "model": "jev-1.13.0", + "answers": {"verdict": {"type": "noul", "noul": 0.9}, "applicable": {"type": "noul", "noul": 0.9}}, + "usage": {"input_tokens": 12, "output_tokens": 0}, + } + with patch("products.ai_observability.backend.llm.system_one.pinned_request", return_value=response) as request: + state, message = Client.validate_key("typesafe", "test-typesafe-key") + + assert state == expected_state + assert (message is None) == (expected_state == "ok") + assert request.call_args.args == ("POST", "https://api.typesafe.ai/v1/systemone") + assert request.call_args.kwargs["headers"]["Authorization"] == "Bearer test-typesafe-key" + + +@pytest.mark.parametrize("allows_na", [False, True]) +def test_typesafe_boolean_request(allows_na: bool) -> None: + response = Mock(status_code=200) + response.json.return_value = { + "model": "jev-1.13.0", + "answers": {"verdict": {"type": "noul", "noul": 0.8}, "applicable": {"type": "noul", "noul": 0.2}}, + "usage": {"input_tokens": 120, "output_tokens": 10}, + } + with patch("products.ai_observability.backend.llm.system_one.pinned_request", return_value=response) as request: + result = SystemOneClient.evaluate_boolean( + api_key="test-typesafe-key", + model="jev-1.13.0", + prompt="Is the response polite?", + source="Hello!", + allows_na=allows_na, + ) + + assert result.answers["verdict"].noul == 0.8 + body = request.call_args.kwargs["json"] + assert body["state"] == "Hello!" + assert body["questions"]["verdict"] == {"type": "noul", "instructions": "Is the response polite?"} + assert ("applicable" in body["questions"]) == allows_na + assert request.call_args.args[1] == "https://api.typesafe.ai/v1/systemone" + + +@pytest.mark.parametrize("probability", [-0.1, 1.1, float("nan"), float("inf"), "0.8", True, None]) +def test_typesafe_rejects_invalid_probabilities(probability: object) -> None: + response = Mock(status_code=200) + response.json.return_value = { + "model": "jev-1.13.0", + "answers": {"verdict": {"type": "noul", "noul": probability}}, + "usage": {"input_tokens": 120, "output_tokens": 10}, + } + with ( + patch("products.ai_observability.backend.llm.system_one.pinned_request", return_value=response), + pytest.raises(StructuredOutputParseError), + ): + SystemOneClient.evaluate_boolean( + api_key="test-typesafe-key", + model="jev-1.13.0", + prompt="Is the response polite?", + source="Hello!", + allows_na=False, + ) + + +@pytest.mark.parametrize("status", [429, 503, 529]) +def test_typesafe_rate_limits_are_retryable(status: int) -> None: + response = Mock(status_code=status, headers={"Retry-After": "15"}) + with ( + patch("products.ai_observability.backend.llm.system_one.pinned_request", return_value=response), + pytest.raises(SystemOneRateLimitError) as error, + ): + SystemOneClient.evaluate_boolean( + api_key="test-typesafe-key", + model="jev-1.13.0", + prompt="Is the response polite?", + source="Hello!", + allows_na=False, + ) + assert error.value.retry_after == 15 + + +def test_typesafe_requires_a_key() -> None: + with ( + patch("products.ai_observability.backend.llm.system_one.pinned_request") as request, + pytest.raises(AuthenticationError), + ): + SystemOneClient.evaluate_boolean( + api_key="", model="jev-1.13.0", prompt="Polite?", source="Hello!", allows_na=False + ) + request.assert_not_called() + + +@pytest.mark.parametrize("answers", [{}, {"verdict": {"type": "noul", "noul": 0.9}}]) +def test_typesafe_requires_every_requested_answer(answers: dict[str, object]) -> None: + response = Mock(status_code=200) + response.json.return_value = { + "model": "jev-1.13.0", + "answers": answers, + "usage": {"input_tokens": 12, "output_tokens": 2}, + } + with ( + patch("products.ai_observability.backend.llm.system_one.pinned_request", return_value=response), + pytest.raises(StructuredOutputParseError), + ): + SystemOneClient.evaluate_boolean( + api_key="test-typesafe-key", model="jev-1.13.0", prompt="Polite?", source="Hello!", allows_na=True + ) + + +@pytest.mark.parametrize( + "status,message,error_type", + [ + (401, "Invalid key", AuthenticationError), + (403, "Access denied", ModelPermissionError), + (500, "Unavailable", ProviderConnectionError), + (413, "Request too large", ContextWindowExceededError), + (422, "Input exceeds the context window", ContextWindowExceededError), + (422, "Invalid question", StructuredOutputParseError), + ], +) +def test_typesafe_preserves_error_categories(status: int, message: str, error_type: type[Exception]) -> None: + response = Mock(status_code=status, text=message) + with ( + patch("products.ai_observability.backend.llm.system_one.pinned_request", return_value=response), + pytest.raises(error_type), + ): + SystemOneClient.evaluate_boolean( + api_key="test-typesafe-key", model="jev-1.13.0", prompt="Polite?", source="Hello!", allows_na=False + ) + + +@pytest.mark.parametrize("api_key", ["example-token", ""]) +def test_custom_endpoint_and_model(api_key: str) -> None: + response = Mock(status_code=200) + response.json.return_value = { + "model": "custom-model-revision", + "answers": {"verdict": {"type": "noul", "noul": 0.7}, "applicable": {"type": "noul", "noul": 1.0}}, + "usage": {"input_tokens": 15, "output_tokens": 0}, + "latency_ms": 42, + } + with patch("products.ai_observability.backend.llm.system_one.pinned_request", return_value=response) as request: + result = SystemOneClient.evaluate_boolean( + api_key=api_key, + base_url="https://decisions.example.com/v1/", + model="custom-model", + prompt="Polite?", + source="Hello!", + allows_na=True, + ) + assert request.call_args.args == ("POST", "https://decisions.example.com/v1/systemone") + assert request.call_args.kwargs["headers"] == ({"Authorization": f"Bearer {api_key}"} if api_key else {}) + assert request.call_args.kwargs["json"]["model"] == "custom-model" + assert result.model == "custom-model-revision" + assert result.usage.output_tokens == 0 + assert result.answers["verdict"].noul == 0.7 + + +@pytest.mark.parametrize( + "base_url", + [ + "http://example.com/v1", + "https://user:secret@example.com/v1", + "https://example.com/v1?token=secret", + "https://example.com/v1#fragment", + ], +) +def test_invalid_endpoint_is_rejected_before_sending_credentials(base_url: str) -> None: + with patch("products.ai_observability.backend.llm.system_one.pinned_request") as request: + state, _ = SystemOneClient.validate_key("example-token", base_url=base_url) + assert state == "error" + request.assert_not_called() + + +def test_private_endpoint_is_blocked() -> None: + with override_settings(DEBUG=False, TEST=False), patch("requests.Session.request") as request: + state, _ = SystemOneClient.validate_key("example-token", base_url="https://127.0.0.1/v1") + assert state == "error" + request.assert_not_called() diff --git a/products/ai_observability/backend/llm/test/test_typesafe.py b/products/ai_observability/backend/llm/test/test_typesafe.py deleted file mode 100644 index 7815eccd0188..000000000000 --- a/products/ai_observability/backend/llm/test/test_typesafe.py +++ /dev/null @@ -1,122 +0,0 @@ -import pytest -from unittest.mock import Mock, patch - -from products.ai_observability.backend.llm.client import Client -from products.ai_observability.backend.llm.errors import ( - AuthenticationError, - ContextWindowExceededError, - ModelPermissionError, - ProviderConnectionError, - StructuredOutputParseError, -) -from products.ai_observability.backend.llm.typesafe import TypeSafeClient, TypeSafeRateLimitError - - -@pytest.mark.parametrize("status, expected_state", [(200, "ok"), (401, "invalid"), (403, "invalid"), (500, "error")]) -def test_typesafe_key_validation(status: int, expected_state: str) -> None: - response = Mock(status_code=status) - response.json.return_value = {"models": [{"name": "jev-latest"}]} - with patch("requests.request", return_value=response) as request: - state, message = Client.validate_key("typesafe", "test-typesafe-key") - - assert state == expected_state - assert (message is None) == (expected_state == "ok") - assert request.call_args.args == ("GET", "https://api.typesafe.ai/v1/models") - assert request.call_args.kwargs["headers"]["Authorization"] == "Bearer test-typesafe-key" - - -@pytest.mark.parametrize("allows_na", [False, True]) -def test_typesafe_boolean_request(allows_na: bool) -> None: - response = Mock(status_code=200) - response.json.return_value = { - "model": "jev-1.13.0", - "answers": {"verdict": {"type": "noul", "noul": 0.8}, "applicable": {"type": "noul", "noul": 0.2}}, - "usage": {"input_tokens": 120, "output_tokens": 10}, - } - with patch("requests.request", return_value=response) as request: - result = TypeSafeClient.evaluate_boolean( - api_key="test-typesafe-key", - model="jev-1.13.0", - prompt="Is the response polite?", - source="Hello!", - allows_na=allows_na, - ) - - assert result.answers["verdict"].noul == 0.8 - body = request.call_args.kwargs["json"] - assert body["state"] == "Hello!" - assert body["questions"]["verdict"] == {"type": "noul", "instructions": "Is the response polite?"} - assert ("applicable" in body["questions"]) == allows_na - assert request.call_args.kwargs["allow_redirects"] is False - - -@pytest.mark.parametrize("probability", [-0.1, 1.1, float("nan"), float("inf"), "0.8", True, None]) -def test_typesafe_rejects_invalid_probabilities(probability: object) -> None: - response = Mock(status_code=200) - response.json.return_value = { - "model": "jev-1.13.0", - "answers": {"verdict": {"type": "noul", "noul": probability}}, - "usage": {"input_tokens": 120, "output_tokens": 10}, - } - with patch("requests.request", return_value=response), pytest.raises(StructuredOutputParseError): - TypeSafeClient.evaluate_boolean( - api_key="test-typesafe-key", - model="jev-1.13.0", - prompt="Is the response polite?", - source="Hello!", - allows_na=False, - ) - - -@pytest.mark.parametrize("status", [429, 529]) -def test_typesafe_rate_limits_are_retryable(status: int) -> None: - response = Mock(status_code=status, headers={"Retry-After": "15"}) - with patch("requests.request", return_value=response), pytest.raises(TypeSafeRateLimitError) as error: - TypeSafeClient.evaluate_boolean( - api_key="test-typesafe-key", - model="jev-1.13.0", - prompt="Is the response polite?", - source="Hello!", - allows_na=False, - ) - assert error.value.retry_after == 15 - - -def test_typesafe_requires_a_key() -> None: - with patch("requests.request") as request, pytest.raises(AuthenticationError): - TypeSafeClient.evaluate_boolean( - api_key="", model="jev-1.13.0", prompt="Polite?", source="Hello!", allows_na=False - ) - request.assert_not_called() - - -@pytest.mark.parametrize("answers", [{}, {"verdict": {"type": "noul", "noul": 0.9}}]) -def test_typesafe_requires_every_requested_answer(answers: dict[str, object]) -> None: - response = Mock(status_code=200) - response.json.return_value = { - "model": "jev-1.13.0", - "answers": answers, - "usage": {"input_tokens": 12, "output_tokens": 2}, - } - with patch("requests.request", return_value=response), pytest.raises(StructuredOutputParseError): - TypeSafeClient.evaluate_boolean( - api_key="test-typesafe-key", model="jev-1.13.0", prompt="Polite?", source="Hello!", allows_na=True - ) - - -@pytest.mark.parametrize( - "status,message,error_type", - [ - (401, "Invalid key", AuthenticationError), - (403, "Access denied", ModelPermissionError), - (500, "Unavailable", ProviderConnectionError), - (422, "Input exceeds the context window", ContextWindowExceededError), - (422, "Invalid question", StructuredOutputParseError), - ], -) -def test_typesafe_preserves_error_categories(status: int, message: str, error_type: type[Exception]) -> None: - response = Mock(status_code=status, text=message) - with patch("requests.request", return_value=response), pytest.raises(error_type): - TypeSafeClient.evaluate_boolean( - api_key="test-typesafe-key", model="jev-1.13.0", prompt="Polite?", source="Hello!", allows_na=False - ) diff --git a/products/ai_observability/backend/llm/typesafe.py b/products/ai_observability/backend/llm/typesafe.py deleted file mode 100644 index a005e3389fab..000000000000 --- a/products/ai_observability/backend/llm/typesafe.py +++ /dev/null @@ -1,120 +0,0 @@ -import math -from datetime import UTC, datetime -from email.utils import parsedate_to_datetime -from typing import Literal - -import requests -from pydantic import BaseModel, Field, ValidationError - -from products.ai_observability.backend.llm.errors import ( - AuthenticationError, - ContextWindowExceededError, - ModelNotFoundError, - ModelPermissionError, - ProviderConnectionError, - RateLimitError, - StructuredOutputParseError, - is_context_window_error_message, -) - - -class NoulAnswer(BaseModel): - type: Literal["noul"] - noul: float = Field(strict=True, ge=0, le=1, allow_inf_nan=False) - - -class TypeSafeUsage(BaseModel): - input_tokens: int = Field(strict=True, ge=0) - output_tokens: int = Field(strict=True, ge=0) - - -class TypeSafeResponse(BaseModel): - model: str = Field(min_length=1) - answers: dict[str, NoulAnswer] - usage: TypeSafeUsage - - -class TypeSafeRateLimitError(RateLimitError): - def __init__(self, retry_after: str | None) -> None: - super().__init__("TypeSafe is temporarily rate limiting requests.") - self.retry_after: float | None = None - if retry_after: - try: - delay = float(retry_after) - except ValueError: - try: - delay = (parsedate_to_datetime(retry_after) - datetime.now(UTC)).total_seconds() - except (ValueError, TypeError, OverflowError): - return - if math.isfinite(delay): - self.retry_after = max(1, min(delay, 300)) - - -class TypeSafeClient: - MODEL = "jev-1.13.0" - - @staticmethod - def evaluate_boolean(*, api_key: str, model: str, prompt: str, source: str, allows_na: bool) -> TypeSafeResponse: - if not api_key: - raise AuthenticationError("A TypeSafe API key is required.") - questions = {"verdict": {"type": "noul", "instructions": prompt}} - if allows_na: - questions["applicable"] = { - "type": "noul", - "instructions": ( - "Do these evaluation criteria apply to this input? Answer true when the criteria can be " - "evaluated, even if they are not met. Answer false only when they are not relevant.\n\n" + prompt - ), - } - try: - response = requests.request( - "POST", - "https://api.typesafe.ai/v1/systemone", - headers={"Authorization": f"Bearer {api_key}"}, - json={"model": model, "state": source, "questions": questions}, - timeout=60, - allow_redirects=False, - ) - except requests.RequestException as error: - raise ProviderConnectionError("Could not reach TypeSafe.") from error - - if response.status_code == 401: - raise AuthenticationError("TypeSafe rejected this API key.") - if response.status_code == 403: - raise ModelPermissionError(model) - if response.status_code == 404: - raise ModelNotFoundError(model) - if response.status_code in (429, 529): - raise TypeSafeRateLimitError(response.headers.get("Retry-After")) - if response.status_code >= 500: - raise ProviderConnectionError("TypeSafe is temporarily unavailable.") - if response.status_code == 422 and is_context_window_error_message(response.text): - raise ContextWindowExceededError("This input exceeds Jev's context window.") - if response.status_code != 200: - raise StructuredOutputParseError("TypeSafe rejected the evaluation request. Check the model and criteria.") - try: - result = TypeSafeResponse.model_validate(response.json()) - except (ValidationError, ValueError) as error: - raise StructuredOutputParseError("TypeSafe returned an invalid evaluation response.") from error - # boffin: Missing answers cannot become a false verdict. - if not questions.keys() <= result.answers.keys(): - raise StructuredOutputParseError("TypeSafe did not answer every evaluation question.") - return result - - @staticmethod - def validate_key(api_key: str) -> tuple[str, str | None]: - try: - response = requests.request( - "GET", - "https://api.typesafe.ai/v1/models", - headers={"Authorization": f"Bearer {api_key}"}, - timeout=30, - allow_redirects=False, - ) - except requests.RequestException: - return "error", "Could not reach TypeSafe. Try again." - if response.status_code == 200: - return "ok", None - if response.status_code in (401, 403): - return "invalid", "TypeSafe rejected this API key. Check the key and try again." - return "error", "Could not validate the TypeSafe key. Try again." diff --git a/products/ai_observability/backend/models/model_configuration.py b/products/ai_observability/backend/models/model_configuration.py index 1c7d199bdc27..7babb8fba51b 100644 --- a/products/ai_observability/backend/models/model_configuration.py +++ b/products/ai_observability/backend/models/model_configuration.py @@ -60,7 +60,7 @@ def get_available_models(self) -> list[str]: from products.ai_observability.backend.llm.client import Client api_key = self.provider_key.encrypted_config.get("api_key") - return Client.list_models(self.provider, api_key) + return Client.list_models(self.provider, api_key, **self.provider_key.provider_extra_kwargs()) from products.ai_observability.backend.llm import PLAYGROUND_MODELS_BY_PROVIDER diff --git a/products/ai_observability/backend/models/provider_keys.py b/products/ai_observability/backend/models/provider_keys.py index b51f27473086..f2077b51f7b1 100644 --- a/products/ai_observability/backend/models/provider_keys.py +++ b/products/ai_observability/backend/models/provider_keys.py @@ -22,7 +22,7 @@ class LLMProvider(models.TextChoices): TOGETHER_AI = "together_ai", "Together AI" MINIMAX = "minimax", "MiniMax" ZEABUR = "zeabur", "Zeabur AI Hub" - TYPESAFE = "typesafe", "TypeSafe" + TYPESAFE = "typesafe", "System One (Jev)" def llm_provider_choices() -> list[tuple[str, str | Promise]]: @@ -67,6 +67,10 @@ def provider_extra_kwargs(self) -> dict[str, Any]: extra config (e.g. Azure's ``azure_endpoint`` and ``api_version``). Most providers return an empty dict. """ + if self.provider == LLMProvider.TYPESAFE: + return { + field: self.encrypted_config[field] for field in ("base_url", "model") if field in self.encrypted_config + } if self.provider == LLMProvider.AZURE_OPENAI: return { "azure_endpoint": self.encrypted_config.get("azure_endpoint", ""), diff --git a/products/ai_observability/frontend/generated/api.schemas.ts b/products/ai_observability/frontend/generated/api.schemas.ts index 7d5dd381ca0d..26145676d499 100644 --- a/products/ai_observability/frontend/generated/api.schemas.ts +++ b/products/ai_observability/frontend/generated/api.schemas.ts @@ -850,7 +850,7 @@ export const EvaluationTargetEnumApi = { * * `together_ai` - Together AI * * `minimax` - MiniMax * * `zeabur` - Zeabur AI Hub - * * `typesafe` - TypeSafe + * * `typesafe` - System One (Jev) */ export type LLMProviderEnumApi = (typeof LLMProviderEnumApi)[keyof typeof LLMProviderEnumApi] @@ -1567,6 +1567,23 @@ export interface LLMProviderKeyApi { readonly error_message: string | null api_key?: string readonly api_key_masked: string + /** System One API base URL, including /v1. Defaults to TypeSafe. */ + base_url?: string + /** + * Model ID served by the System One endpoint. Defaults to jev-1.13.0. + * @maxLength 100 + */ + system_one_model?: string + /** + * Configured System One base URL. + * @nullable + */ + readonly base_url_display: string | null + /** + * Configured System One model ID. + * @nullable + */ + readonly system_one_model_display: string | null /** Azure OpenAI endpoint URL */ azure_endpoint?: string /** @@ -2103,6 +2120,23 @@ export interface PatchedLLMProviderKeyApi { readonly error_message?: string | null api_key?: string readonly api_key_masked?: string + /** System One API base URL, including /v1. Defaults to TypeSafe. */ + base_url?: string + /** + * Model ID served by the System One endpoint. Defaults to jev-1.13.0. + * @maxLength 100 + */ + system_one_model?: string + /** + * Configured System One base URL. + * @nullable + */ + readonly base_url_display?: string | null + /** + * Configured System One model ID. + * @nullable + */ + readonly system_one_model_display?: string | null /** Azure OpenAI endpoint URL */ azure_endpoint?: string /** diff --git a/products/ai_observability/frontend/generated/api.zod.ts b/products/ai_observability/frontend/generated/api.zod.ts index 5fab62024e2b..4caf45710557 100644 --- a/products/ai_observability/frontend/generated/api.zod.ts +++ b/products/ai_observability/frontend/generated/api.zod.ts @@ -441,7 +441,7 @@ export const EvaluationsCreateBody = /* @__PURE__ */ zod 'typesafe', ]) .describe( - '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub\n\* `typesafe` - TypeSafe' + '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub\n\* `typesafe` - System One (Jev)' ), model: zod.string().max(evaluationsCreateBodyModelConfigurationOneModelMax), provider_key_id: zod @@ -746,7 +746,7 @@ export const EvaluationsUpdateBody = /* @__PURE__ */ zod 'typesafe', ]) .describe( - '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub\n\* `typesafe` - TypeSafe' + '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub\n\* `typesafe` - System One (Jev)' ), model: zod.string().max(evaluationsUpdateBodyModelConfigurationOneModelMax), provider_key_id: zod @@ -955,7 +955,7 @@ export const EvaluationsPartialUpdateBody = /* @__PURE__ */ zod 'typesafe', ]) .describe( - '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub\n\* `typesafe` - TypeSafe' + '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub\n\* `typesafe` - System One (Jev)' ), model: zod.string().max(evaluationsPartialUpdateBodyModelConfigurationOneModelMax), provider_key_id: zod @@ -1517,6 +1517,8 @@ export const LlmAnalyticsParserRecipesPartialUpdateBody = /* @__PURE__ */ zod.ob export const llmAnalyticsProviderKeysCreateBodyNameMax = 255 +export const llmAnalyticsProviderKeysCreateBodySystemOneModelMax = 100 + export const llmAnalyticsProviderKeysCreateBodyApiVersionMax = 20 export const llmAnalyticsProviderKeysCreateBodySetAsActiveDefault = false @@ -1536,10 +1538,16 @@ export const LlmAnalyticsProviderKeysCreateBody = /* @__PURE__ */ zod.object({ 'typesafe', ]) .describe( - '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub\n\* `typesafe` - TypeSafe' + '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub\n\* `typesafe` - System One (Jev)' ), name: zod.string().max(llmAnalyticsProviderKeysCreateBodyNameMax), api_key: zod.string().optional(), + base_url: zod.url().optional().describe('System One API base URL, including \/v1. Defaults to TypeSafe.'), + system_one_model: zod + .string() + .max(llmAnalyticsProviderKeysCreateBodySystemOneModelMax) + .optional() + .describe('Model ID served by the System One endpoint. Defaults to jev-1.13.0.'), azure_endpoint: zod.url().optional().describe('Azure OpenAI endpoint URL'), api_version: zod .string() @@ -1551,6 +1559,8 @@ export const LlmAnalyticsProviderKeysCreateBody = /* @__PURE__ */ zod.object({ export const llmAnalyticsProviderKeysUpdateBodyNameMax = 255 +export const llmAnalyticsProviderKeysUpdateBodySystemOneModelMax = 100 + export const llmAnalyticsProviderKeysUpdateBodyApiVersionMax = 20 export const llmAnalyticsProviderKeysUpdateBodySetAsActiveDefault = false @@ -1570,10 +1580,16 @@ export const LlmAnalyticsProviderKeysUpdateBody = /* @__PURE__ */ zod.object({ 'typesafe', ]) .describe( - '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub\n\* `typesafe` - TypeSafe' + '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub\n\* `typesafe` - System One (Jev)' ), name: zod.string().max(llmAnalyticsProviderKeysUpdateBodyNameMax), api_key: zod.string().optional(), + base_url: zod.url().optional().describe('System One API base URL, including \/v1. Defaults to TypeSafe.'), + system_one_model: zod + .string() + .max(llmAnalyticsProviderKeysUpdateBodySystemOneModelMax) + .optional() + .describe('Model ID served by the System One endpoint. Defaults to jev-1.13.0.'), azure_endpoint: zod.url().optional().describe('Azure OpenAI endpoint URL'), api_version: zod .string() @@ -1585,6 +1601,8 @@ export const LlmAnalyticsProviderKeysUpdateBody = /* @__PURE__ */ zod.object({ export const llmAnalyticsProviderKeysPartialUpdateBodyNameMax = 255 +export const llmAnalyticsProviderKeysPartialUpdateBodySystemOneModelMax = 100 + export const llmAnalyticsProviderKeysPartialUpdateBodyApiVersionMax = 20 export const llmAnalyticsProviderKeysPartialUpdateBodySetAsActiveDefault = false @@ -1605,10 +1623,16 @@ export const LlmAnalyticsProviderKeysPartialUpdateBody = /* @__PURE__ */ zod.obj ]) .optional() .describe( - '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub\n\* `typesafe` - TypeSafe' + '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub\n\* `typesafe` - System One (Jev)' ), name: zod.string().max(llmAnalyticsProviderKeysPartialUpdateBodyNameMax).optional(), api_key: zod.string().optional(), + base_url: zod.url().optional().describe('System One API base URL, including \/v1. Defaults to TypeSafe.'), + system_one_model: zod + .string() + .max(llmAnalyticsProviderKeysPartialUpdateBodySystemOneModelMax) + .optional() + .describe('Model ID served by the System One endpoint. Defaults to jev-1.13.0.'), azure_endpoint: zod.url().optional().describe('Azure OpenAI endpoint URL'), api_version: zod .string() @@ -1620,6 +1644,8 @@ export const LlmAnalyticsProviderKeysPartialUpdateBody = /* @__PURE__ */ zod.obj export const llmAnalyticsProviderKeysValidateCreateBodyNameMax = 255 +export const llmAnalyticsProviderKeysValidateCreateBodySystemOneModelMax = 100 + export const llmAnalyticsProviderKeysValidateCreateBodyApiVersionMax = 20 export const llmAnalyticsProviderKeysValidateCreateBodySetAsActiveDefault = false @@ -1639,10 +1665,16 @@ export const LlmAnalyticsProviderKeysValidateCreateBody = /* @__PURE__ */ zod.ob 'typesafe', ]) .describe( - '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub\n\* `typesafe` - TypeSafe' + '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub\n\* `typesafe` - System One (Jev)' ), name: zod.string().max(llmAnalyticsProviderKeysValidateCreateBodyNameMax), api_key: zod.string().optional(), + base_url: zod.url().optional().describe('System One API base URL, including \/v1. Defaults to TypeSafe.'), + system_one_model: zod + .string() + .max(llmAnalyticsProviderKeysValidateCreateBodySystemOneModelMax) + .optional() + .describe('Model ID served by the System One endpoint. Defaults to jev-1.13.0.'), azure_endpoint: zod.url().optional().describe('Azure OpenAI endpoint URL'), api_version: zod .string() diff --git a/products/ai_observability/frontend/modelPickerLogic.test.ts b/products/ai_observability/frontend/modelPickerLogic.test.ts index ede2501f8805..d997dd864bc8 100644 --- a/products/ai_observability/frontend/modelPickerLogic.test.ts +++ b/products/ai_observability/frontend/modelPickerLogic.test.ts @@ -45,39 +45,42 @@ describe('modelPickerLogic', () => { }) describe('loadByokModels', () => { - it('offers Jev only to evaluation model pickers', async () => { - useMocks({ - get: { - '/api/environments/:team_id/llm_analytics/provider_keys/': { - results: [{ id: 'key-typesafe', provider: 'typesafe', name: 'TypeSafe', state: 'ok' }], + it.each(['jev-1.13.0', 'custom-model'])( + 'offers System One model %s only to evaluation model pickers', + async (model) => { + useMocks({ + get: { + '/api/environments/:team_id/llm_analytics/provider_keys/': { + results: [{ id: 'key-typesafe', provider: 'typesafe', name: 'TypeSafe', state: 'ok' }], + }, + '/api/environments/:team_id/llm_analytics/evaluation_config/': { active_provider_key: null }, + '/api/llm_proxy/models/': ({ request }) => + new URL(request.url).searchParams.get('provider_key_id') + ? [ + 200, + [ + { + id: model, + name: model, + provider: 'System One (Jev)', + is_recommended: true, + }, + ], + ] + : [200, []], }, - '/api/environments/:team_id/llm_analytics/evaluation_config/': { active_provider_key: null }, - '/api/llm_proxy/models/': ({ request }) => - new URL(request.url).searchParams.get('provider_key_id') - ? [ - 200, - [ - { - id: 'jev-1.13.0', - name: 'jev-1.13.0', - provider: 'TypeSafe', - is_recommended: true, - }, - ], - ] - : [200, []], - }, - }) - logic = modelPickerLogic() - logic.mount() - await expectLogic(logic).toFinishAllListeners() - - expect(logic.values.evaluationProviderModelGroups[0].models[0].id).toBe('jev-1.13.0') - expect(logic.values.providerModelGroups).toEqual([]) - expect(logic.values.generativeByokModels).toEqual([]) - expect(logic.values.hasByokKeys).toBe(false) - expect(logic.values.evaluationModelNotice).toBeNull() - }) + }) + logic = modelPickerLogic() + logic.mount() + await expectLogic(logic).toFinishAllListeners() + + expect(logic.values.evaluationProviderModelGroups[0].models[0].id).toBe(model) + expect(logic.values.providerModelGroups).toEqual([]) + expect(logic.values.generativeByokModels).toEqual([]) + expect(logic.values.hasByokKeys).toBe(false) + expect(logic.values.evaluationModelNotice).toBeNull() + } + ) it('should load and attach providerKeyId to models from valid keys', async () => { useMocks({ diff --git a/products/ai_observability/frontend/settings/LLMProviderKeysSettings.tsx b/products/ai_observability/frontend/settings/LLMProviderKeysSettings.tsx index 2d581f873f1e..8e04f248b65d 100644 --- a/products/ai_observability/frontend/settings/LLMProviderKeysSettings.tsx +++ b/products/ai_observability/frontend/settings/LLMProviderKeysSettings.tsx @@ -24,6 +24,8 @@ import { AlternativeKey, CreateLLMProviderKeyPayload, DEFAULT_AZURE_API_VERSION, + DEFAULT_SYSTEM_ONE_BASE_URL, + DEFAULT_SYSTEM_ONE_MODEL, DependentConfigsResponse, KeyValidationResult, LLMProvider, @@ -34,6 +36,7 @@ import { llmProviderKeysLogic, sortProviderKeys, } from './llmProviderKeysLogic' +import { SystemOneConnectionFields } from './SystemOneConnectionFields' function StateTag({ state, errorMessage }: { state: LLMProviderKeyState; errorMessage: string | null }): JSX.Element { const tagProps: { type: 'success' | 'danger' | 'warning' | 'default'; children: string } = { @@ -103,7 +106,7 @@ function getKeyPlaceholder(provider: LLMProvider): string { case 'zeabur': return 'sk-...' case 'typesafe': - return 'Enter your TypeSafe API key' + return 'Enter your endpoint’s bearer token' } } @@ -134,7 +137,11 @@ function KeyValidationStatus({ const bullets = (
  • Your key will be encrypted and stored securely
  • -
  • You pay {LLM_PROVIDER_LABELS[provider]} directly for model usage
  • +
  • + {provider === 'typesafe' + ? 'Model usage is billed by your endpoint provider' + : `You pay ${LLM_PROVIDER_LABELS[provider]} directly for model usage`} +
  • Each evaluation counts as an AI observability event
) @@ -158,6 +165,7 @@ function KeyValidationStatus({ } function AddKeyModal({ restrictionReason }: { restrictionReason: string | null }): JSX.Element { + const { systemOneBaseUrl, systemOneModel } = useValues(llmProviderKeysLogic) const { newKeyModalOpen, providerKeysLoading, preValidationResult, preValidationResultLoading, evaluationConfig } = useValues(llmProviderKeysLogic) const { setNewKeyModalOpen, createProviderKey, preValidateKey, clearPreValidation } = @@ -172,7 +180,15 @@ function AddKeyModal({ restrictionReason }: { restrictionReason: string | null } const isAzure = provider === 'azure_openai' const keyValidated = preValidationResult?.state === 'ok' - const isValid = name.length > 0 && apiKey.length > 0 && (!isAzure || azureEndpoint.length > 0) + const isSystemOne = provider === 'typesafe' + const isValid = + name.length > 0 && + (isSystemOne + ? systemOneBaseUrl.length > 0 && + systemOneModel.length > 0 && + (apiKey.length > 0 || systemOneBaseUrl !== DEFAULT_SYSTEM_ONE_BASE_URL) + : apiKey.length > 0) && + (!isAzure || azureEndpoint.length > 0) const validationFailed = !!preValidationResult && preValidationResult.state !== 'ok' const azureErrorField = isAzure && validationFailed ? azureErrorFieldFromResult(preValidationResult) : null @@ -226,12 +242,24 @@ function AddKeyModal({ restrictionReason }: { restrictionReason: string | null } } const handleSubmit = (): void => { + if (isSystemOne) { + createProviderKey({ + payload: { + provider, + name, + api_key: apiKey, + base_url: systemOneBaseUrl, + system_one_model: systemOneModel, + }, + }) + return + } if (keyValidated) { const payload: CreateLLMProviderKeyPayload = { provider, name, api_key: apiKey, - set_as_active: provider !== 'typesafe' && !evaluationConfig?.active_provider_key, + set_as_active: !evaluationConfig?.active_provider_key, } if (isAzure) { payload.azure_endpoint = azureEndpoint @@ -249,6 +277,9 @@ function AddKeyModal({ restrictionReason }: { restrictionReason: string | null } } const handleApiKeyBlur = (): void => { + if (isSystemOne) { + return + } if (apiKey.length > 0 && !preValidationResult) { preValidateKey({ apiKey, @@ -287,7 +318,7 @@ function AddKeyModal({ restrictionReason }: { restrictionReason: string | null } @@ -344,6 +375,7 @@ function AddKeyModal({ restrictionReason }: { restrictionReason: string | null }
)} + {isSystemOne && }
+ {isSystemOne && ( +

+ Sent as a bearer token. Leave empty only if your custom endpoint does not require + authentication. +

+ )}
@@ -388,8 +426,11 @@ function EditKeyModal({ restrictionReason: string | null }): JSX.Element { const { providerKeysLoading, preValidationResult, preValidationResultLoading } = useValues(llmProviderKeysLogic) + const { systemOneBaseUrl, systemOneModel } = useValues(llmProviderKeysLogic) const { setEditingKey, updateProviderKey, preValidateKey, clearPreValidation } = useActions(llmProviderKeysLogic) const isAzureEdit = keyToEdit.provider === 'azure_openai' + const isSystemOne = keyToEdit.provider === 'typesafe' + const endpointChanged = systemOneBaseUrl !== (keyToEdit.base_url_display ?? DEFAULT_SYSTEM_ONE_BASE_URL) const [name, setName] = useState(keyToEdit.name) const [apiKey, setApiKey] = useState('') @@ -409,6 +450,15 @@ function EditKeyModal({ if (apiKey.length > 0) { payload.api_key = apiKey } + if (isSystemOne) { + if (endpointChanged) { + payload.base_url = systemOneBaseUrl + payload.api_key = apiKey + } + if (systemOneModel !== (keyToEdit.system_one_model_display ?? DEFAULT_SYSTEM_ONE_MODEL)) { + payload.system_one_model = systemOneModel + } + } if (isAzureEdit) { if (azureEndpoint !== (keyToEdit.azure_endpoint_display ?? '')) { payload.azure_endpoint = azureEndpoint @@ -421,6 +471,9 @@ function EditKeyModal({ } const handleApiKeyBlur = (): void => { + if (isSystemOne) { + return + } if (apiKey.length > 0) { preValidateKey({ apiKey, @@ -437,8 +490,9 @@ function EditKeyModal({ } } - const keyValidated = apiKey.length === 0 || preValidationResult?.state === 'ok' - const isValid = name.length > 0 && keyValidated + const keyValidated = isSystemOne || apiKey.length === 0 || preValidationResult?.state === 'ok' + const isValid = + name.length > 0 && keyValidated && (!isSystemOne || (systemOneBaseUrl.length > 0 && systemOneModel.length > 0)) const validationFailed = !!preValidationResult && preValidationResult.state !== 'ok' const azureErrorField = isAzureEdit && validationFailed ? azureErrorFieldFromResult(preValidationResult) : null @@ -502,6 +556,7 @@ function EditKeyModal({ )} + {isSystemOne && }
@@ -512,7 +567,11 @@ function EditKeyModal({ value={apiKey} onChange={handleApiKeyChange} onBlur={handleApiKeyBlur} - placeholder={`Leave empty to keep current (${keyToEdit.api_key_masked})`} + placeholder={ + isSystemOne && endpointChanged + ? 'Enter the new endpoint’s bearer token' + : `Leave empty to keep current (${keyToEdit.api_key_masked})` + } type="password" autoComplete="off" className="mt-1" @@ -527,7 +586,11 @@ function EditKeyModal({ suppressError={azureErrorField === 'endpoint'} /> ) : ( -

Leave empty to keep the current key

+

+ {isSystemOne && endpointChanged + ? 'The saved key will not be sent to the new endpoint. Enter its bearer token, or leave empty for no authentication.' + : 'Leave empty to keep the current key'} +

)}
diff --git a/products/ai_observability/frontend/settings/SystemOneConnectionFields.stories.tsx b/products/ai_observability/frontend/settings/SystemOneConnectionFields.stories.tsx new file mode 100644 index 000000000000..ba7483ff4310 --- /dev/null +++ b/products/ai_observability/frontend/settings/SystemOneConnectionFields.stories.tsx @@ -0,0 +1,28 @@ +import type { Meta, StoryObj } from '@storybook/react' + +import { useStorybookMocks } from '~/mocks/browser' + +import { LLMProviderKeysSettings } from './LLMProviderKeysSettings' +import { SystemOneConnectionFields } from './SystemOneConnectionFields' + +const meta: Meta = { + title: 'Scenes-App/AI observability/System One connection', + component: SystemOneConnectionFields, + decorators: [ + (Story) => { + useStorybookMocks({ + get: { + '/api/environments/:team_id/llm_analytics/provider_keys/': { results: [] }, + '/api/environments/:team_id/llm_analytics/evaluation_config/': { active_provider_key: null }, + }, + }) + return + }, + ], +} + +export default meta +type Story = StoryObj + +export const Default: Story = {} +export const ProviderSettings: Story = { render: () => } diff --git a/products/ai_observability/frontend/settings/SystemOneConnectionFields.tsx b/products/ai_observability/frontend/settings/SystemOneConnectionFields.tsx new file mode 100644 index 000000000000..a80dc7022be7 --- /dev/null +++ b/products/ai_observability/frontend/settings/SystemOneConnectionFields.tsx @@ -0,0 +1,48 @@ +import { useActions, useValues } from 'kea' + +import { LemonInput } from '@posthog/lemon-ui' + +import { LemonLabel } from 'lib/lemon-ui/LemonLabel' + +import { llmProviderKeysLogic } from './llmProviderKeysLogic' + +export function SystemOneConnectionFields(): JSX.Element { + const { systemOneBaseUrl, systemOneModel, providerKeysLoading } = useValues(llmProviderKeysLogic) + const { setSystemOneBaseUrl, setSystemOneModel } = useActions(llmProviderKeysLogic) + + return ( +
+ Advanced configuration +
+
+ Base URL + +
+
+ Model ID + +
+

+ Use a public HTTPS endpoint that supports the System One API. Include /v1 in the base URL if + required by your service. Validation sends a short synthetic example to the selected model. +

+
+
+ ) +} diff --git a/products/ai_observability/frontend/settings/llmProviderKeysLogic.test.ts b/products/ai_observability/frontend/settings/llmProviderKeysLogic.test.ts index ba250b022182..4bbcd45411ac 100644 --- a/products/ai_observability/frontend/settings/llmProviderKeysLogic.test.ts +++ b/products/ai_observability/frontend/settings/llmProviderKeysLogic.test.ts @@ -17,6 +17,8 @@ describe('normalizeLLMProvider', () => { ['zeabur-ai-hub', 'zeabur'], ['zeabur', 'zeabur'], ['openai', 'openai'], + ['System One (Jev)', 'typesafe'], + ['typesafe', 'typesafe'], ])('maps %s to %s', (input, expected) => { expect(normalizeLLMProvider(input)).toBe(expected) }) diff --git a/products/ai_observability/frontend/settings/llmProviderKeysLogic.ts b/products/ai_observability/frontend/settings/llmProviderKeysLogic.ts index 369dfdf1cb00..d35e8ea9d2ef 100644 --- a/products/ai_observability/frontend/settings/llmProviderKeysLogic.ts +++ b/products/ai_observability/frontend/settings/llmProviderKeysLogic.ts @@ -5,13 +5,15 @@ import api, { ApiError } from 'lib/api' import { lemonToast } from 'lib/lemon-ui/LemonToast/LemonToast' import { teamLogic } from 'scenes/teamLogic' -import type { LLMProviderEnumApi } from '../generated/api.schemas' +import type { LLMProviderEnumApi, LLMProviderKeyApi } from '../generated/api.schemas' export type LLMProviderKeyState = 'unknown' | 'ok' | 'invalid' | 'error' export type LLMProvider = LLMProviderEnumApi /** Default Azure OpenAI API version — keep in sync with backend DEFAULT_API_VERSION. */ export const DEFAULT_AZURE_API_VERSION = '2024-10-21' +export const DEFAULT_SYSTEM_ONE_BASE_URL: string = 'https://api.typesafe.ai/v1' +export const DEFAULT_SYSTEM_ONE_MODEL: string = 'jev-1.13.0' export const LLM_PROVIDER_LABELS: Record = { openai: 'OpenAI', @@ -23,7 +25,7 @@ export const LLM_PROVIDER_LABELS: Record = { together_ai: 'Together AI', minimax: 'MiniMax', zeabur: 'Zeabur AI Hub', - typesafe: 'TypeSafe', + typesafe: 'System One (Jev)', } const LLM_PROVIDERS = new Set(Object.keys(LLM_PROVIDER_LABELS)) @@ -57,6 +59,9 @@ export function normalizeLLMProvider(provider: string | undefined): LLMProvider } const normalized = provider.trim().toLowerCase() + if (normalized === 'system one (jev)') { + return 'typesafe' + } if (normalized === 'google' || normalized === 'google-ai-studio') { return 'gemini' } @@ -76,7 +81,9 @@ export function normalizeLLMProvider(provider: string | undefined): LLMProvider return normalized in LLM_PROVIDER_LABELS ? (normalized as LLMProvider) : null } -export interface LLMProviderKey { +export interface LLMProviderKey extends Partial< + Pick +> { id: string provider: LLMProvider name: string @@ -138,7 +145,7 @@ export interface EvaluationConfig { updated_at: string } -export interface CreateLLMProviderKeyPayload { +export interface CreateLLMProviderKeyPayload extends Pick { provider: LLMProvider name: string api_key: string @@ -147,7 +154,7 @@ export interface CreateLLMProviderKeyPayload { api_version?: string } -export interface UpdateLLMProviderKeyPayload { +export interface UpdateLLMProviderKeyPayload extends Pick { name?: string api_key?: string azure_endpoint?: string @@ -194,6 +201,8 @@ export interface llmProviderKeysLogicValues { providerKeys: LLMProviderKey[] providerKeysLoading: boolean requiresProviderKey: boolean + systemOneBaseUrl: string + systemOneModel: string validatingKeyId: string | null } @@ -350,6 +359,12 @@ export interface llmProviderKeysLogicActions { setNewKeyModalOpen: (open: boolean) => { open: boolean } + setSystemOneBaseUrl: (baseUrl: string) => { + baseUrl: string + } + setSystemOneModel: (model: string) => { + model: string + } updateProviderKey: ({ id, payload }: { id: string; payload: UpdateLLMProviderKeyPayload }) => { id: string payload: UpdateLLMProviderKeyPayload @@ -416,6 +431,8 @@ export const llmProviderKeysLogic = kea([ path(['products', 'ai_observability', 'settings', 'llmProviderKeysLogic']), actions({ + setSystemOneBaseUrl: (baseUrl: string) => ({ baseUrl }), + setSystemOneModel: (model: string) => ({ model }), clearPreValidation: true, setNewKeyModalOpen: (open: boolean) => ({ open }), setEditingKey: (key: LLMProviderKey | null) => ({ key }), @@ -424,6 +441,22 @@ export const llmProviderKeysLogic = kea([ }), reducers({ + systemOneBaseUrl: [ + DEFAULT_SYSTEM_ONE_BASE_URL, + { + setSystemOneBaseUrl: (_, { baseUrl }) => baseUrl, + setNewKeyModalOpen: () => DEFAULT_SYSTEM_ONE_BASE_URL, + setEditingKey: (_, { key }) => key?.base_url_display ?? DEFAULT_SYSTEM_ONE_BASE_URL, + }, + ], + systemOneModel: [ + DEFAULT_SYSTEM_ONE_MODEL, + { + setSystemOneModel: (_, { model }) => model, + setNewKeyModalOpen: () => DEFAULT_SYSTEM_ONE_MODEL, + setEditingKey: (_, { key }) => key?.system_one_model_display ?? DEFAULT_SYSTEM_ONE_MODEL, + }, + ], newKeyModalOpen: [ false, { diff --git a/services/mcp/src/api/generated.ts b/services/mcp/src/api/generated.ts index 8bf9eccedcdc..35421d7ff314 100644 --- a/services/mcp/src/api/generated.ts +++ b/services/mcp/src/api/generated.ts @@ -36259,7 +36259,7 @@ export namespace Schemas { * * `together_ai` - Together AI * * `minimax` - MiniMax * * `zeabur` - Zeabur AI Hub - * * `typesafe` - TypeSafe + * * `typesafe` - System One (Jev) */ export type LLMProviderEnum = typeof LLMProviderEnum[keyof typeof LLMProviderEnum]; @@ -36477,6 +36477,23 @@ export namespace Schemas { readonly error_message: string | null; api_key?: string; readonly api_key_masked: string; + /** System One API base URL, including /v1. Defaults to TypeSafe. */ + base_url?: string; + /** + * Model ID served by the System One endpoint. Defaults to jev-1.13.0. + * @maxLength 100 + */ + system_one_model?: string; + /** + * Configured System One base URL. + * @nullable + */ + readonly base_url_display: string | null; + /** + * Configured System One model ID. + * @nullable + */ + readonly system_one_model_display: string | null; /** Azure OpenAI endpoint URL */ azure_endpoint?: string; /** @@ -70055,6 +70072,23 @@ export namespace Schemas { readonly error_message?: string | null; api_key?: string; readonly api_key_masked?: string; + /** System One API base URL, including /v1. Defaults to TypeSafe. */ + base_url?: string; + /** + * Model ID served by the System One endpoint. Defaults to jev-1.13.0. + * @maxLength 100 + */ + system_one_model?: string; + /** + * Configured System One base URL. + * @nullable + */ + readonly base_url_display?: string | null; + /** + * Configured System One model ID. + * @nullable + */ + readonly system_one_model_display?: string | null; /** Azure OpenAI endpoint URL */ azure_endpoint?: string; /** diff --git a/services/mcp/src/generated/ai_observability/api.ts b/services/mcp/src/generated/ai_observability/api.ts index d3bae257e86b..2c2e077da6cb 100644 --- a/services/mcp/src/generated/ai_observability/api.ts +++ b/services/mcp/src/generated/ai_observability/api.ts @@ -743,7 +743,7 @@ export const EvaluationsCreateBody = () => zod 'typesafe', ]) .describe( - '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub\n\* `typesafe` - TypeSafe' + '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub\n\* `typesafe` - System One (Jev)' ), model: zod.string().max(evaluationsCreateBodyModelConfigurationOneModelMax), provider_key_id: zod @@ -970,7 +970,7 @@ export const EvaluationsPartialUpdateBody = () => zod 'typesafe', ]) .describe( - '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub\n\* `typesafe` - TypeSafe' + '\* `openai` - Openai\n\* `anthropic` - Anthropic\n\* `gemini` - Gemini\n\* `openrouter` - Openrouter\n\* `fireworks` - Fireworks\n\* `azure_openai` - Azure OpenAI\n\* `together_ai` - Together AI\n\* `minimax` - MiniMax\n\* `zeabur` - Zeabur AI Hub\n\* `typesafe` - System One (Jev)' ), model: zod.string().max(evaluationsPartialUpdateBodyModelConfigurationOneModelMax), provider_key_id: zod From 6f46380ffcb2bf71b29f10e78c76bdb0d8d947e9 Mon Sep 17 00:00:00 2001 From: Bernat Torres Date: Fri, 25 Sep 2026 12:58:32 +0200 Subject: [PATCH 04/30] fix(aio): distinguish rejected System One requests --- .../internal/ai-observability-judge-inputs.md | 1 + .../ai_observability/evaluation_errors.py | 8 ++++++ .../ai_observability/evaluation_llm_judge.py | 16 ++++++++++- .../ai_observability/test_run_evaluation.py | 27 +++++++++++++++++++ .../backend/llm/system_one.py | 10 +++++-- .../backend/llm/test/test_system_one.py | 8 ++++-- 6 files changed, 65 insertions(+), 5 deletions(-) diff --git a/docs/internal/ai-observability-judge-inputs.md b/docs/internal/ai-observability-judge-inputs.md index ea1dae6944b1..ad90c30e5006 100644 --- a/docs/internal/ai-observability-judge-inputs.md +++ b/docs/internal/ai-observability-judge-inputs.md @@ -65,6 +65,7 @@ Jev provides no written reasoning, so reports inspect the original source when e Rate limits and overload responses are retried through Temporal, honoring `Retry-After` up to five minutes. If retries fail, the run fails and the evaluation stays enabled. +Blocked endpoints and rejected requests disable the evaluation and mark the connection for revalidation, without recording model usage. Invalid probabilities or missing answers fail the evaluation rather than producing a false result. Inputs rejected for exceeding the model's context window are skipped. See TypeSafe's [API reference](https://docs.typesafe.ai/api) and [model limits and pricing](https://docs.typesafe.ai/models). diff --git a/posthog/temporal/ai_observability/evaluation_errors.py b/posthog/temporal/ai_observability/evaluation_errors.py index 6b70bf5182c7..884cfa425bd1 100644 --- a/posthog/temporal/ai_observability/evaluation_errors.py +++ b/posthog/temporal/ai_observability/evaluation_errors.py @@ -31,6 +31,14 @@ class EvaluationErrorSpec: USER_ERROR_SPECS: dict[str, EvaluationErrorSpec] = { + "request_rejected": EvaluationErrorSpec( + error_type="request_rejected", + owner="user", + safe_message="The judge endpoint rejected the request. Check the connection settings and evaluation criteria.", + status_reason=EvaluationStatusReason.PROVIDER_KEY_INVALID, + disables_evaluation=True, + provider_key_state=LLMProviderKey.State.ERROR, + ), "provider_key_required": EvaluationErrorSpec( error_type="provider_key_required", owner="user", diff --git a/posthog/temporal/ai_observability/evaluation_llm_judge.py b/posthog/temporal/ai_observability/evaluation_llm_judge.py index 7b0102c5d7bd..684ae7c67c60 100644 --- a/posthog/temporal/ai_observability/evaluation_llm_judge.py +++ b/posthog/temporal/ai_observability/evaluation_llm_judge.py @@ -48,7 +48,11 @@ RateLimitError, StructuredOutputParseError, ) -from products.ai_observability.backend.llm.system_one import SystemOneClient, SystemOneRateLimitError +from products.ai_observability.backend.llm.system_one import ( + SystemOneClient, + SystemOneRateLimitError, + SystemOneRequestRejectedError, +) from products.ai_observability.backend.llm.types import CompletionResponse from products.ai_observability.backend.models.evaluation_configs import NumericOutputConfig, NumericScoreOutOfBounds from products.ai_observability.backend.text_repr.formatters import add_line_numbers, reduce_by_uniform_sampling @@ -504,6 +508,16 @@ def call_llm_judge( response_format=response_format, ) ) + except SystemOneRequestRejectedError as e: + increment_user_errors("request_rejected", provider=provider) + return terminal_user_error_result( + spec=require_user_error_spec("request_rejected", is_byok=is_byok), + message=str(e), + allows_na=allows_na, + output_type=output_type, + key_id=key_id, + is_byok=is_byok, + ) except SystemOneRateLimitError as e: increment_errors("rate_limit", provider=provider) raise ApplicationError( diff --git a/posthog/temporal/ai_observability/test_run_evaluation.py b/posthog/temporal/ai_observability/test_run_evaluation.py index 963d07f1eb93..b6d7d3630cc3 100644 --- a/posthog/temporal/ai_observability/test_run_evaluation.py +++ b/posthog/temporal/ai_observability/test_run_evaluation.py @@ -158,6 +158,33 @@ def test_typesafe_judge_emits_boolean_probability_without_reasoning( assert properties["$ai_evaluation_key_type"] == "byok" +@pytest.mark.parametrize("status", [301, 400, 422]) +def test_system_one_rejected_requests_disable_without_model_cost_attribution(status: int) -> None: + key = MagicMock(provider="typesafe", encrypted_config={"api_key": "example-token"}) + with ( + patch("posthog.temporal.ai_observability.evaluation_llm_judge.model_spec") as spec, + patch( + "products.ai_observability.backend.llm.system_one.pinned_request", + return_value=MagicMock(status_code=status, text="Invalid request"), + ), + ): + spec.return_value.resolve.return_value = MagicMock( + provider="typesafe", model="jev-1.13.0", provider_key=key, is_byok=True + ) + result = call_llm_judge( + evaluation={"id": "test-evaluation", "team_id": 1, "evaluation_config": {"prompt": "Polite?"}}, + system_prompt="", + user_prompt="Hello!", + allows_na=False, + ) + assert result["terminal_user_error"] is True + assert result["skip_reason"] == "request_rejected" + assert result["provider_key_state"] == "error" + assert result["status_reason"] == "provider_key_invalid" + assert "model" not in result + assert "provider" not in result + + def test_typesafe_rate_limit_retries_without_disabling_the_evaluation() -> None: key = MagicMock(provider="typesafe", encrypted_config={"api_key": "test-typesafe-key"}) with ( diff --git a/products/ai_observability/backend/llm/system_one.py b/products/ai_observability/backend/llm/system_one.py index 76b74da2dca8..d797c6f819f9 100644 --- a/products/ai_observability/backend/llm/system_one.py +++ b/products/ai_observability/backend/llm/system_one.py @@ -13,6 +13,7 @@ from products.ai_observability.backend.llm.errors import ( AuthenticationError, ContextWindowExceededError, + LLMError, ModelNotFoundError, ModelPermissionError, ProviderConnectionError, @@ -38,6 +39,10 @@ class SystemOneResponse(BaseModel): usage: SystemOneUsage +class SystemOneRequestRejectedError(LLMError): + pass + + class SystemOneRateLimitError(RateLimitError): def __init__(self, retry_after: str | None) -> None: super().__init__("The System One endpoint is busy. Try again later.") @@ -98,7 +103,7 @@ def evaluate_boolean( timeout=60, ) except SSRFBlockedError as error: - raise StructuredOutputParseError("This endpoint is not allowed. Use a public HTTPS endpoint.") from error + raise SystemOneRequestRejectedError("This endpoint is not allowed. Use a public HTTPS endpoint.") from error except requests.RequestException as error: raise ProviderConnectionError("Could not reach the System One endpoint. Try again.") from error @@ -117,7 +122,7 @@ def evaluate_boolean( ): raise ContextWindowExceededError("This input exceeds the endpoint's size limit. Reduce the input.") if response.status_code != 200: - raise StructuredOutputParseError( + raise SystemOneRequestRejectedError( "The endpoint rejected the evaluation request. Check the model and criteria." ) try: @@ -152,6 +157,7 @@ def validate_key(api_key: str, *, base_url: str = BASE_URL, model: str = MODEL) ProviderConnectionError, RateLimitError, StructuredOutputParseError, + SystemOneRequestRejectedError, ContextWindowExceededError, ) as error: return "error", str(error) diff --git a/products/ai_observability/backend/llm/test/test_system_one.py b/products/ai_observability/backend/llm/test/test_system_one.py index 1f56f2cad602..f9d0b998d3c7 100644 --- a/products/ai_observability/backend/llm/test/test_system_one.py +++ b/products/ai_observability/backend/llm/test/test_system_one.py @@ -11,7 +11,11 @@ ProviderConnectionError, StructuredOutputParseError, ) -from products.ai_observability.backend.llm.system_one import SystemOneClient, SystemOneRateLimitError +from products.ai_observability.backend.llm.system_one import ( + SystemOneClient, + SystemOneRateLimitError, + SystemOneRequestRejectedError, +) @pytest.mark.parametrize("status, expected_state", [(200, "ok"), (401, "invalid"), (403, "invalid"), (500, "error")]) @@ -130,7 +134,7 @@ def test_typesafe_requires_every_requested_answer(answers: dict[str, object]) -> (500, "Unavailable", ProviderConnectionError), (413, "Request too large", ContextWindowExceededError), (422, "Input exceeds the context window", ContextWindowExceededError), - (422, "Invalid question", StructuredOutputParseError), + (422, "Invalid question", SystemOneRequestRejectedError), ], ) def test_typesafe_preserves_error_categories(status: int, message: str, error_type: type[Exception]) -> None: From 25623e9ba2b7ee74e196256153f82f71a4e5b88d Mon Sep 17 00:00:00 2001 From: "tests-posthog[bot]" <250237707+tests-posthog[bot]@users.noreply.github.com> Date: Fri, 25 Sep 2026 11:04:12 +0000 Subject: [PATCH 05/30] test(mcp): update unit test snapshots --- .../unit/__snapshots__/tool-schemas/llma-evaluation-create.json | 2 +- .../unit/__snapshots__/tool-schemas/llma-evaluation-update.json | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/services/mcp/tests/unit/__snapshots__/tool-schemas/llma-evaluation-create.json b/services/mcp/tests/unit/__snapshots__/tool-schemas/llma-evaluation-create.json index 4b8de8d2ab63..c7acc821350a 100644 --- a/services/mcp/tests/unit/__snapshots__/tool-schemas/llma-evaluation-create.json +++ b/services/mcp/tests/unit/__snapshots__/tool-schemas/llma-evaluation-create.json @@ -112,7 +112,7 @@ "type": "string" }, "provider": { - "description": "* `openai` - Openai\n* `anthropic` - Anthropic\n* `gemini` - Gemini\n* `openrouter` - Openrouter\n* `fireworks` - Fireworks\n* `azure_openai` - Azure OpenAI\n* `together_ai` - Together AI\n* `minimax` - MiniMax\n* `zeabur` - Zeabur AI Hub\n* `typesafe` - TypeSafe", + "description": "* `openai` - Openai\n* `anthropic` - Anthropic\n* `gemini` - Gemini\n* `openrouter` - Openrouter\n* `fireworks` - Fireworks\n* `azure_openai` - Azure OpenAI\n* `together_ai` - Together AI\n* `minimax` - MiniMax\n* `zeabur` - Zeabur AI Hub\n* `typesafe` - System One (Jev)", "enum": [ "openai", "anthropic", diff --git a/services/mcp/tests/unit/__snapshots__/tool-schemas/llma-evaluation-update.json b/services/mcp/tests/unit/__snapshots__/tool-schemas/llma-evaluation-update.json index 004588151d45..69dc4c521d7a 100644 --- a/services/mcp/tests/unit/__snapshots__/tool-schemas/llma-evaluation-update.json +++ b/services/mcp/tests/unit/__snapshots__/tool-schemas/llma-evaluation-update.json @@ -115,7 +115,7 @@ "type": "string" }, "provider": { - "description": "* `openai` - Openai\n* `anthropic` - Anthropic\n* `gemini` - Gemini\n* `openrouter` - Openrouter\n* `fireworks` - Fireworks\n* `azure_openai` - Azure OpenAI\n* `together_ai` - Together AI\n* `minimax` - MiniMax\n* `zeabur` - Zeabur AI Hub\n* `typesafe` - TypeSafe", + "description": "* `openai` - Openai\n* `anthropic` - Anthropic\n* `gemini` - Gemini\n* `openrouter` - Openrouter\n* `fireworks` - Fireworks\n* `azure_openai` - Azure OpenAI\n* `together_ai` - Together AI\n* `minimax` - MiniMax\n* `zeabur` - Zeabur AI Hub\n* `typesafe` - System One (Jev)", "enum": [ "openai", "anthropic", From 46d304558501ed7429e464d4268ff1780b8ea198 Mon Sep 17 00:00:00 2001 From: Bernat Torres Date: Fri, 25 Sep 2026 13:43:32 +0200 Subject: [PATCH 06/30] feat(aio): support typed system one questions and numeric judges --- .../internal/ai-observability-judge-inputs.md | 29 ++- .../eval_reports/report_agent/prompts.py | 2 +- .../ai_observability/evaluation_llm_judge.py | 73 ++++++-- .../ai_observability/test_run_evaluation.py | 67 +++++++ .../backend/api/evaluations.py | 21 ++- .../ai_observability/backend/api/proxy.py | 2 +- .../backend/api/test/test_evaluations.py | 13 +- .../backend/llm/system_one.py | 109 ++++++++--- .../backend/llm/test/test_system_one.py | 175 +++++++++++++++--- .../backend/models/evaluation_configs.py | 18 ++ .../backend/models/provider_keys.py | 2 +- .../models/test/test_evaluation_configs.py | 12 ++ .../components/EvaluationExplanation.tsx | 2 +- .../evaluations/AIObservabilityEvaluation.tsx | 26 ++- .../NumericEvaluationConfig.stories.tsx | 7 +- .../NumericEvaluationConfig.test.tsx | 17 ++ .../components/NumericEvaluationConfig.tsx | 35 +++- .../frontend/evaluations/constants.ts | 18 +- .../evaluations/llmEvaluationLogic.test.ts | 32 ++-- .../evaluations/llmEvaluationLogic.ts | 8 +- .../frontend/generated/api.schemas.ts | 42 ++++- .../frontend/generated/api.zod.ts | 64 ++++++- .../frontend/modelPickerLogic.test.ts | 2 +- .../settings/SystemOneConnectionFields.tsx | 5 +- .../settings/llmProviderKeysLogic.test.ts | 1 + .../frontend/settings/llmProviderKeysLogic.ts | 4 +- services/mcp/src/api/generated.ts | 42 ++++- .../mcp/src/generated/ai_observability/api.ts | 41 +++- 28 files changed, 730 insertions(+), 139 deletions(-) diff --git a/docs/internal/ai-observability-judge-inputs.md b/docs/internal/ai-observability-judge-inputs.md index ad90c30e5006..876436fedac5 100644 --- a/docs/internal/ai-observability-judge-inputs.md +++ b/docs/internal/ai-observability-judge-inputs.md @@ -39,34 +39,45 @@ They sample the combined input, tool definitions, and output only when that text Implementation: [trace judge](../../posthog/temporal/ai_observability/run_trace_evaluation.py), [session judge](../../posthog/temporal/ai_observability/run_session_evaluation.py), and [generation judge](../../posthog/temporal/ai_observability/evaluation_llm_judge.py). -## System One boolean judges +## System One judges System One-compatible models are available under the existing LLM judge option. -Add a connection under **System One (Jev)** in provider key settings. +Add a connection under **System One** in provider key settings. The default endpoint is TypeSafe's `https://api.typesafe.ai/v1`, with model `jev-1.13.0` and a TypeSafe API key. Advanced configuration accepts a different public HTTPS base URL and model ID for compatible services. The client appends `/systemone` to the base URL and sends the API key as a bearer token. An empty key selects no authentication for a custom endpoint; TypeSafe requires a key. Changing the endpoint requires entering its credential again, or explicitly choosing no authentication, so an existing key is not forwarded to a new host. Private network destinations and redirects are blocked by the shared DNS-pinned HTTP transport. -Saving a connection validates it with a short synthetic input and two Noul questions, including applicability, without sending evaluation data. +Saving a connection validates it with a short synthetic input and a Noul question, without sending evaluation data. Select the connection and configured model on each evaluation; these connections cannot become the shared active provider key used by other AI features. Provider keys keep the provider they were created with; switching providers requires a new key. -This integration supports boolean evaluations only and uses the same formatted text for generation, trace, and session targets. +This integration supports boolean and numeric evaluations and uses the same formatted text for generation, trace, and session targets. +The client accepts typed Noul, Score, and Choice questions independently of PostHog's evaluation output types. +Choice support in the client does not enable categorical evaluations in the product. API compatibility does not guarantee equivalent judgments or calibration across models. Compare results on representative inputs when changing models. -The evaluation prompt becomes a [Noul question](https://docs.typesafe.ai/primitives/noul). +For boolean evaluations, the prompt becomes a [Noul question](https://docs.typesafe.ai/primitives/noul). A probability of at least 0.5 produces `true`; the evaluation's existing pass/fail polarity still applies. -For evaluations that allow N/A, a separate question checks whether the criteria apply, using the same threshold. -Uncertainty alone does not produce N/A. The raw probability is stored in `$ai_evaluation_probability`, with token usage and the resolved model version. -Jev provides no written reasoning, so reports inspect the original source when explaining outcomes. + +Numeric evaluations use a [Score question](https://docs.typesafe.ai/primitives/score). +Configure 2 to 10 ordered score levels and a minimum and maximum. +The levels are evenly spaced across those bounds. +The response is a probability-weighted level index, which maps to the configured scale without rounding. +For example, three levels on a 0–10 scale map an answer of 1.25 to a score of 6.25. +Scores are stored in `$ai_evaluation_numeric_result`; they are not boolean probabilities. +The rubric remains part of the prompt when switching to a completion-based judge. + +For either output type, evaluations that allow N/A send a separate Noul question about whether the criteria apply, using the 0.5 threshold. +Uncertainty alone does not produce N/A. +System One answers contain no written reasoning, so reports inspect the original source when explaining outcomes. Rate limits and overload responses are retried through Temporal, honoring `Retry-After` up to five minutes. If retries fail, the run fails and the evaluation stays enabled. Blocked endpoints and rejected requests disable the evaluation and mark the connection for revalidation, without recording model usage. -Invalid probabilities or missing answers fail the evaluation rather than producing a false result. +Invalid probabilities, missing answers, mismatched answer types, and invalid score scales skip the item as an unparsable response. Inputs rejected for exceeding the model's context window are skipped. See TypeSafe's [API reference](https://docs.typesafe.ai/api) and [model limits and pricing](https://docs.typesafe.ai/models). diff --git a/posthog/temporal/ai_observability/eval_reports/report_agent/prompts.py b/posthog/temporal/ai_observability/eval_reports/report_agent/prompts.py index c67d70be2797..4163b72f598c 100644 --- a/posthog/temporal/ai_observability/eval_reports/report_agent/prompts.py +++ b/posthog/temporal/ai_observability/eval_reports/report_agent/prompts.py @@ -170,7 +170,7 @@ def build_eval_report_system_prompt( sample_ordering_signature = "" sample_ordering_instruction = ( 'Rows include full reasoning when available. Use the default `order_by="recent"`. ' - "Some judges, including Jev, return no written reasoning. For those results, inspect the original " + "Some judges return no written reasoning. For those results, inspect the original " "generation, trace, or session with the detail tools and ground your analysis in that source. " "Do not invent a judge explanation or treat absent reasoning as an evaluation failure." ) diff --git a/posthog/temporal/ai_observability/evaluation_llm_judge.py b/posthog/temporal/ai_observability/evaluation_llm_judge.py index 684ae7c67c60..b331a9148c7b 100644 --- a/posthog/temporal/ai_observability/evaluation_llm_judge.py +++ b/posthog/temporal/ai_observability/evaluation_llm_judge.py @@ -49,7 +49,12 @@ StructuredOutputParseError, ) from products.ai_observability.backend.llm.system_one import ( + NoulAnswer, + NoulQuestion, + ScoreAnswer, + ScoreQuestion, SystemOneClient, + SystemOneQuestion, SystemOneRateLimitError, SystemOneRequestRejectedError, ) @@ -175,6 +180,11 @@ def get_output_type_config( instructions += f" The score must be at most {numeric_config.max}." if numeric_config.step is not None: instructions += f" Suggested score increment: {numeric_config.step}; do not round an otherwise valid score." + if numeric_config.score_levels is not None: + instructions += " Use these rubric levels, interpolating between them when needed:\n" + "\n".join( + f"{numeric_config.score_from_level(index)}: {description}" + for index, description in enumerate(numeric_config.score_levels) + ) if allows_na: instructions += " Return score=null when the criteria does not apply." return OutputTypeConfig( @@ -444,8 +454,8 @@ def call_llm_judge( raise provider = resolved.provider - if provider == "typesafe" and output_type != "boolean": - raise ApplicationError("System One connections support boolean evaluations only.", non_retryable=True) + if provider == "typesafe" and output_type not in ("boolean", "numeric"): + raise ApplicationError("This System One evaluation output type is not supported.", non_retryable=True) model = resolved.model provider_key = resolved.provider_key is_byok = resolved.is_byok @@ -467,26 +477,59 @@ def call_llm_judge( if provider == "typesafe": if provider_key is not None and provider_key.provider != provider: raise ProviderMismatchError(provider_key.provider, provider) - system_one_result = SystemOneClient.evaluate_boolean( + prompt = evaluation["evaluation_config"]["prompt"] + numeric_config = NumericOutputConfig.model_validate(output_config) if output_type == "numeric" else None + questions: dict[str, SystemOneQuestion] + if numeric_config is not None: + if numeric_config.score_levels is None: + raise ApplicationError("Add score levels to this numeric evaluation.", non_retryable=True) + questions = {"score": ScoreQuestion(instructions=prompt, criteria=list(numeric_config.score_levels))} + else: + questions = {"verdict": NoulQuestion(instructions=prompt)} + if allows_na: + questions["applicable"] = NoulQuestion( + instructions=( + "Do these evaluation criteria apply to this input? Answer true when the criteria can be " + "evaluated, even if they are not met. Answer false only when they are not relevant.\n\n" + + prompt + ) + ) + system_one_result = SystemOneClient.evaluate( api_key=provider_key.encrypted_config.get("api_key", "") if provider_key else "", base_url=provider_key.encrypted_config.get("base_url", SystemOneClient.BASE_URL) if provider_key else SystemOneClient.BASE_URL, model=model, - prompt=evaluation["evaluation_config"]["prompt"], - source=user_prompt, - allows_na=allows_na, + state=user_prompt, + questions=questions, ) - probability = system_one_result.answers["verdict"].noul - applicable = not allows_na or system_one_result.answers["applicable"].noul >= 0.5 - parsed: BooleanEvalResult | BooleanWithNAEvalResult = ( - BooleanWithNAEvalResult( - reasoning="", - outcome="not_applicable" if not applicable else "pass" if probability >= 0.5 else "fail", + applicable = True + if allows_na: + applicability_answer = system_one_result.answers["applicable"] + assert isinstance(applicability_answer, NoulAnswer) + applicable = applicability_answer.noul >= 0.5 + parsed: BooleanEvalResult | BooleanWithNAEvalResult | NumericEvalResult | NumericWithNAEvalResult + if numeric_config is not None: + score_answer = system_one_result.answers["score"] + assert isinstance(score_answer, ScoreAnswer) + score = numeric_config.score_from_level(score_answer.score) + parsed = ( + NumericWithNAEvalResult(reasoning="", score=score if applicable else None) + if allows_na + else NumericEvalResult(reasoning="", score=score) + ) + else: + verdict_answer = system_one_result.answers["verdict"] + assert isinstance(verdict_answer, NoulAnswer) + probability = verdict_answer.noul + parsed = ( + BooleanWithNAEvalResult( + reasoning="", + outcome="not_applicable" if not applicable else "pass" if probability >= 0.5 else "fail", + ) + if allows_na + else BooleanEvalResult(reasoning="", verdict=probability >= 0.5) ) - if allows_na - else BooleanEvalResult(reasoning="", verdict=probability >= 0.5) - ) model = system_one_result.model response = CompletionResponse( content="", diff --git a/posthog/temporal/ai_observability/test_run_evaluation.py b/posthog/temporal/ai_observability/test_run_evaluation.py index b6d7d3630cc3..cf84b6dad99c 100644 --- a/posthog/temporal/ai_observability/test_run_evaluation.py +++ b/posthog/temporal/ai_observability/test_run_evaluation.py @@ -158,6 +158,73 @@ def test_typesafe_judge_emits_boolean_probability_without_reasoning( assert properties["$ai_evaluation_key_type"] == "byok" +@pytest.mark.parametrize( + "raw_score,probabilities,allows_na,applicability,expected", + [ + (0, [1, 0, 0], False, 1, -10), + (1, [0, 1, 0], False, 1, 0), + (1.25, [0.1, 0.55, 0.35], True, 1, 2.5), + (2, [0, 0, 1], True, 0.1, None), + ], +) +def test_system_one_numeric_scores_keep_the_rubric_scale( + raw_score: float, probabilities: list[float], allows_na: bool, applicability: float, expected: float | None +) -> None: + levels = ["Poor", "Fair", "Good"] + evaluation = { + "id": "test-evaluation", + "name": "Quality", + "team_id": 1, + "evaluation_config": {"prompt": "Assess response quality"}, + "output_type": "numeric", + "output_config": {"min": -10, "max": 10, "score_levels": levels, "allows_na": allows_na}, + } + key = MagicMock(provider="typesafe", encrypted_config={"api_key": "example-token"}) + response = MagicMock(status_code=200) + response.json.return_value = { + "model": "jev-1.13.0", + "answers": { + "score": { + "type": "score", + "score": raw_score, + "confidence": 0.5, + "legend": {str(index): value for index, value in enumerate(levels)}, + "probabilities": {str(index): value for index, value in enumerate(probabilities)}, + }, + "applicable": {"type": "noul", "noul": applicability}, + }, + "usage": {"input_tokens": 120, "output_tokens": 10}, + } + with ( + patch("posthog.temporal.ai_observability.evaluation_llm_judge.model_spec") as spec, + patch("products.ai_observability.backend.llm.system_one.pinned_request", return_value=response) as request, + ): + spec.return_value.resolve.return_value = MagicMock( + provider="typesafe", model="jev-1.13.0", provider_key=key, is_byok=True + ) + result = call_llm_judge(evaluation=evaluation, system_prompt="", user_prompt="Hello!", allows_na=allows_na) + assert request.call_args.kwargs["json"]["questions"]["score"] == { + "type": "score", + "instructions": "Assess response quality", + "criteria": levels, + } + assert ("applicable" in request.call_args.kwargs["json"]["questions"]) is allows_na + assert result.get("score") == expected + if expected is None: + assert "score" not in result + assert result["applicable"] is False + assert result["reasoning"] == "" + assert result["total_tokens"] == 130 + assert "verdict" not in result + assert "probability" not in result + properties = build_evaluation_event_properties(evaluation, result, datetime.now(UTC)) + assert properties["$ai_evaluation_result_type"] == "numeric" + assert properties.get("$ai_evaluation_numeric_result") == expected + assert "$ai_evaluation_result" not in properties + assert "$ai_evaluation_probability" not in properties + assert properties["$ai_model"] == "jev-1.13.0" + + @pytest.mark.parametrize("status", [301, 400, 422]) def test_system_one_rejected_requests_disable_without_model_cost_attribution(status: int) -> None: key = MagicMock(provider="typesafe", encrypted_config={"api_key": "example-token"}) diff --git a/products/ai_observability/backend/api/evaluations.py b/products/ai_observability/backend/api/evaluations.py index f4e51c928b3a..e146aab155d6 100644 --- a/products/ai_observability/backend/api/evaluations.py +++ b/products/ai_observability/backend/api/evaluations.py @@ -195,6 +195,14 @@ class _EvaluationConfigField(serializers.JSONField): "minimum": 0, "description": "Optional positive input increment. Does not round evaluation results.", }, + "score_levels": { + "type": "array", + "nullable": True, + "items": {"type": "string", "minLength": 1}, + "minItems": 2, + "maxItems": 10, + "description": "Ordered rubric descriptions, spaced evenly from min to max. Required for System One numeric judges. Both bounds must be set.", + }, "passing_rule": { "type": "object", "nullable": True, @@ -400,7 +408,7 @@ class EvaluationSerializer(UserAccessControlSerializerMixin, serializers.ModelSe help_text=( "Output config. For 'boolean' output_type: {allows_na} to permit N/A results, and " "{true_is_failure} to declare that a true result means the evaluation found a problem. " - "For 'numeric': only min/max/step, allows_na, and passing_rule {operator: 'gte'|'lte', threshold}. " + "For 'numeric': min/max/step, allows_na, score_levels, and passing_rule {operator: 'gte'|'lte', threshold}. " "Do not send true_is_failure for numeric output. For 'sentiment': {}." ), ) @@ -516,9 +524,9 @@ def validate(self, data): if isinstance(model_configuration, dict) else getattr(model_configuration, "provider", None) ) - if model_provider == LLMProvider.TYPESAFE and output_type != "boolean": + if model_provider == LLMProvider.TYPESAFE and output_type not in ("boolean", "numeric"): raise serializers.ValidationError( - {"model_configuration": "System One connections support boolean evaluations only."} + {"model_configuration": "Select a model that supports this evaluation output type."} ) if not evaluation_uses_model_configuration(evaluation_type) and model_configuration is not None: @@ -564,6 +572,13 @@ def validate(self, data): except ValueError as e: raise serializers.ValidationError({"config": str(e)}) + if model_provider == LLMProvider.TYPESAFE and output_type == "numeric": + numeric_output = data.get("output_config", getattr(self.instance, "output_config", {})) + if not numeric_output.get("score_levels"): + raise serializers.ValidationError( + {"output_config": "Add 2 to 10 score levels and set both bounds for this System One judge."} + ) + # Sentiment is addressed per-message within one generation event ($ai_target_event_id + # message index). An aggregate target emits a single evaluation event for the whole unit, # where the message index is ambiguous and that per-generation linkage is absent. diff --git a/products/ai_observability/backend/api/proxy.py b/products/ai_observability/backend/api/proxy.py index 77bee39e080c..73f09dad9f1f 100644 --- a/products/ai_observability/backend/api/proxy.py +++ b/products/ai_observability/backend/api/proxy.py @@ -67,7 +67,7 @@ def models_cache_key(provider_key_id: str | uuid.UUID) -> str: PROVIDER_DISPLAY_NAMES: dict[str, str] = { - "typesafe": "System One (Jev)", + "typesafe": "System One", "openai": "OpenAI", "anthropic": "Anthropic", "gemini": "Gemini", diff --git a/products/ai_observability/backend/api/test/test_evaluations.py b/products/ai_observability/backend/api/test/test_evaluations.py index f3ee8f9e232e..537d3739f044 100644 --- a/products/ai_observability/backend/api/test/test_evaluations.py +++ b/products/ai_observability/backend/api/test/test_evaluations.py @@ -108,16 +108,21 @@ def test_existing_boolean_cannot_change_to_numeric(self): class TestModelConfigurationSerializer(SimpleTestCase): - def test_numeric_evaluation_rejects_system_one_connection(self) -> None: + @parameterized.expand([({}, False), ({"min": 0, "max": 10, "score_levels": ["Poor", "Good"]}, True)]) + def test_numeric_system_one_requires_score_levels(self, config: dict, valid: bool) -> None: evaluation = Evaluation( evaluation_type="llm_judge", evaluation_config={"prompt": "Score quality"}, output_type="numeric", - output_config={}, + output_config=config, ) serializer = EvaluationSerializer(instance=evaluation, partial=True) - with self.assertRaisesMessage(ValidationError, "System One connections support boolean evaluations only"): - serializer.validate({"model_configuration": {"provider": "typesafe", "model": "custom-model"}}) + data = {"model_configuration": {"provider": "typesafe", "model": "custom-model"}} + if valid: + self.assertEqual(serializer.validate(data)["model_configuration"], data["model_configuration"]) + else: + with self.assertRaisesMessage(ValidationError, "Add 2 to 10 score levels"): + serializer.validate(data) @parameterized.expand( [ diff --git a/products/ai_observability/backend/llm/system_one.py b/products/ai_observability/backend/llm/system_one.py index d797c6f819f9..51579efa5c61 100644 --- a/products/ai_observability/backend/llm/system_one.py +++ b/products/ai_observability/backend/llm/system_one.py @@ -1,11 +1,12 @@ import math +from collections.abc import Mapping from datetime import UTC, datetime from email.utils import parsedate_to_datetime -from typing import Literal +from typing import Annotated, Literal from urllib.parse import urlsplit import requests -from pydantic import BaseModel, Field, ValidationError +from pydantic import BaseModel, Field, JsonValue, ValidationError from posthog.security.pinned_requests import SSRFBlockedError, pinned_request from posthog.security.url_validation import has_authority_bypass_chars @@ -22,10 +23,49 @@ is_context_window_error_message, ) +type SystemOneContent = str | dict[str, JsonValue] | list[JsonValue] +type Probability = Annotated[float, Field(strict=True, ge=0, le=1, allow_inf_nan=False)] + + +class NoulQuestion(BaseModel): + type: Literal["noul"] = "noul" + instructions: SystemOneContent | None = None + criteria: dict[Literal["true", "false"], SystemOneContent | None] | None = None + + +class ScoreQuestion(BaseModel): + type: Literal["score"] = "score" + instructions: SystemOneContent | None = None + criteria: list[SystemOneContent] = Field(min_length=1, max_length=10) + + +class ChoiceQuestion(BaseModel): + type: Literal["choice"] = "choice" + instructions: SystemOneContent | None = None + criteria: dict[str, SystemOneContent | None] = Field(min_length=1, max_length=255) + + +type SystemOneQuestion = NoulQuestion | ScoreQuestion | ChoiceQuestion + class NoulAnswer(BaseModel): type: Literal["noul"] - noul: float = Field(strict=True, ge=0, le=1, allow_inf_nan=False) + noul: Probability + + +class ScoreAnswer(BaseModel): + type: Literal["score"] + score: float = Field(strict=True, ge=0, allow_inf_nan=False) + legend: dict[str, SystemOneContent] + probabilities: dict[str, Probability] + confidence: Probability + + +class ChoiceAnswer(BaseModel): + type: Literal["choice"] + choice: str + probabilities: dict[str, Probability] + confidence: Probability class SystemOneUsage(BaseModel): @@ -35,7 +75,7 @@ class SystemOneUsage(BaseModel): class SystemOneResponse(BaseModel): model: str = Field(min_length=1) - answers: dict[str, NoulAnswer] + answers: dict[str, Annotated[NoulAnswer | ScoreAnswer | ChoiceAnswer, Field(discriminator="type")]] usage: SystemOneUsage @@ -79,27 +119,31 @@ def normalize_base_url(base_url: str) -> str: return base_url.rstrip("/") @staticmethod - def evaluate_boolean( - *, api_key: str, model: str, prompt: str, source: str, allows_na: bool, base_url: str = BASE_URL + def evaluate( + *, + api_key: str, + model: str, + state: SystemOneContent, + questions: Mapping[str, SystemOneQuestion], + base_url: str = BASE_URL, ) -> SystemOneResponse: base_url = SystemOneClient.normalize_base_url(base_url) if not api_key and base_url == SystemOneClient.BASE_URL: raise AuthenticationError("A TypeSafe API key is required.") - questions = {"verdict": {"type": "noul", "instructions": prompt}} - if allows_na: - questions["applicable"] = { - "type": "noul", - "instructions": ( - "Do these evaluation criteria apply to this input? Answer true when the criteria can be " - "evaluated, even if they are not met. Answer false only when they are not relevant.\n\n" + prompt - ), - } + if not questions: + raise ValueError("Provide at least one System One question.") try: response = pinned_request( "POST", f"{base_url}/systemone", headers={"Authorization": f"Bearer {api_key}"} if api_key else {}, - json={"model": model, "state": source, "questions": questions}, + json={ + "model": model, + "state": state, + "questions": { + key: question.model_dump(mode="json", exclude_none=True) for key, question in questions.items() + }, + }, timeout=60, ) except SSRFBlockedError as error: @@ -131,23 +175,46 @@ def evaluate_boolean( raise StructuredOutputParseError( "The endpoint returned an invalid System One response. Check compatibility." ) from error - # boffin: Missing answers cannot become a false verdict. + # Missing or mismatched answers cannot become valid evaluation results. if not questions.keys() <= result.answers.keys(): raise StructuredOutputParseError( "The endpoint did not answer every evaluation question. Check compatibility." ) + for key, question in questions.items(): + answer = result.answers[key] + if answer.type != question.type: + raise StructuredOutputParseError("The endpoint returned the wrong answer type. Check compatibility.") + if isinstance(question, ChoiceQuestion) and isinstance(answer, ChoiceAnswer): + if set(answer.probabilities) != set(question.criteria) or answer.choice not in question.criteria: + raise StructuredOutputParseError("The endpoint returned different choices. Check compatibility.") + if isinstance(question, ScoreQuestion) and isinstance(answer, ScoreAnswer): + levels = {str(index) for index in range(len(question.criteria))} + if ( + set(answer.probabilities) != levels + or set(answer.legend) != levels + or answer.score > len(levels) - 1 + ): + raise StructuredOutputParseError( + "The endpoint returned an invalid score scale. Check compatibility." + ) + # Compatible servers round each probability to four decimal places. + if isinstance(answer, ScoreAnswer | ChoiceAnswer) and not math.isclose( + sum(answer.probabilities.values()), 1.0, abs_tol=max(0.001, len(answer.probabilities) * 0.00005) + ): + raise StructuredOutputParseError( + "The endpoint returned an invalid probability distribution. Check compatibility." + ) return result @staticmethod def validate_key(api_key: str, *, base_url: str = BASE_URL, model: str = MODEL) -> tuple[str, str | None]: try: - SystemOneClient.evaluate_boolean( + SystemOneClient.evaluate( api_key=api_key, base_url=base_url, model=model, - prompt="Does the text contain a greeting?", - source="Hello!", - allows_na=True, + state="Hello!", + questions={"verdict": NoulQuestion(instructions="Does the text contain a greeting?")}, ) except (AuthenticationError, ModelPermissionError) as error: return "invalid", str(error) diff --git a/products/ai_observability/backend/llm/test/test_system_one.py b/products/ai_observability/backend/llm/test/test_system_one.py index f9d0b998d3c7..4cb03bf4b674 100644 --- a/products/ai_observability/backend/llm/test/test_system_one.py +++ b/products/ai_observability/backend/llm/test/test_system_one.py @@ -12,7 +12,14 @@ StructuredOutputParseError, ) from products.ai_observability.backend.llm.system_one import ( + ChoiceAnswer, + ChoiceQuestion, + NoulAnswer, + NoulQuestion, + ScoreAnswer, + ScoreQuestion, SystemOneClient, + SystemOneQuestion, SystemOneRateLimitError, SystemOneRequestRejectedError, ) @@ -35,31 +42,135 @@ def test_typesafe_key_validation(status: int, expected_state: str) -> None: assert request.call_args.kwargs["headers"]["Authorization"] == "Bearer test-typesafe-key" -@pytest.mark.parametrize("allows_na", [False, True]) -def test_typesafe_boolean_request(allows_na: bool) -> None: +def test_typesafe_mixed_question_request() -> None: response = Mock(status_code=200) response.json.return_value = { "model": "jev-1.13.0", - "answers": {"verdict": {"type": "noul", "noul": 0.8}, "applicable": {"type": "noul", "noul": 0.2}}, + "answers": { + "verdict": {"type": "noul", "noul": 0.8}, + "quality": { + "type": "score", + "score": 1.25, + "legend": {"0": "Poor", "1": "Fair", "2": "Good"}, + "probabilities": {"0": 0.1, "1": 0.55, "2": 0.35}, + "confidence": 0.6, + }, + "language": {"type": "choice", "choice": "en", "probabilities": {"en": 0.7, "es": 0.3}, "confidence": 0.4}, + }, "usage": {"input_tokens": 120, "output_tokens": 10}, } with patch("products.ai_observability.backend.llm.system_one.pinned_request", return_value=response) as request: - result = SystemOneClient.evaluate_boolean( + result = SystemOneClient.evaluate( api_key="test-typesafe-key", model="jev-1.13.0", - prompt="Is the response polite?", - source="Hello!", - allows_na=allows_na, + state={"text": "Hello!"}, + questions={ + "verdict": NoulQuestion(instructions="Is the response polite?"), + "quality": ScoreQuestion(instructions="Assess quality", criteria=["Poor", "Fair", "Good"]), + "language": ChoiceQuestion( + instructions={"question": "Which language?"}, criteria={"en": None, "es": None} + ), + }, ) + assert isinstance(result.answers["verdict"], NoulAnswer) assert result.answers["verdict"].noul == 0.8 + assert isinstance(result.answers["quality"], ScoreAnswer) + assert result.answers["quality"].score == 1.25 + assert isinstance(result.answers["language"], ChoiceAnswer) + assert result.answers["language"].choice == "en" body = request.call_args.kwargs["json"] - assert body["state"] == "Hello!" + assert body["state"] == {"text": "Hello!"} assert body["questions"]["verdict"] == {"type": "noul", "instructions": "Is the response polite?"} - assert ("applicable" in body["questions"]) == allows_na + assert body["questions"]["quality"] == { + "type": "score", + "instructions": "Assess quality", + "criteria": ["Poor", "Fair", "Good"], + } + assert body["questions"]["language"] == { + "type": "choice", + "instructions": {"question": "Which language?"}, + "criteria": {"en": None, "es": None}, + } assert request.call_args.args[1] == "https://api.typesafe.ai/v1/systemone" +@pytest.mark.parametrize( + "question,answer", + [ + (ScoreQuestion(criteria=["Poor", "Good"]), {"type": "noul", "noul": 0.8}), + *[ + ( + ScoreQuestion(criteria=["Poor", "Good"]), + { + "type": "score", + "score": 0.75, + "confidence": 0.5, + "legend": {"0": "Poor", "1": "Good"}, + "probabilities": {"0": 0.25, "1": 0.75}, + field: value, + }, + ) + for field, value in [ + ("score", 1.1), + ("score", True), + ("score", float("nan")), + ("legend", {"0": "Poor"}), + ("probabilities", {"0": 0.25, "2": 0.75}), + ("probabilities", {"0": 0.25, "1": 0.25}), + ] + ], + ( + ChoiceQuestion(criteria={"en": None, "es": None}), + {"type": "choice", "choice": "fr", "confidence": 0.5, "probabilities": {"en": 0.25, "es": 0.75}}, + ), + ], +) +def test_rejects_answers_that_do_not_match_the_requested_scale( + question: SystemOneQuestion, answer: dict[str, object] +) -> None: + response = Mock(status_code=200) + response.json.return_value = { + "model": "custom-model", + "answers": {"result": answer}, + "usage": {"input_tokens": 12, "output_tokens": 2}, + } + with ( + patch("products.ai_observability.backend.llm.system_one.pinned_request", return_value=response), + pytest.raises(StructuredOutputParseError), + ): + SystemOneClient.evaluate( + api_key="example-token", model="custom-model", state="Hello!", questions={"result": question} + ) + + +def test_choice_accepts_a_rounded_distribution_with_many_options() -> None: + criteria = {str(index): None for index in range(255)} + response = Mock(status_code=200) + response.json.return_value = { + "model": "custom-model", + "answers": { + "result": { + "type": "choice", + "choice": "0", + "confidence": 0, + "probabilities": dict.fromkeys(criteria, 0.0039), + } + }, + "usage": {"input_tokens": 12, "output_tokens": 0}, + } + with patch("products.ai_observability.backend.llm.system_one.pinned_request", return_value=response): + result = SystemOneClient.evaluate( + api_key="", + base_url="https://decisions.example.com/v1", + model="custom-model", + state="Hello!", + questions={"result": ChoiceQuestion(criteria=criteria)}, + ) + assert isinstance(result.answers["result"], ChoiceAnswer) + assert result.answers["result"].choice == "0" + + @pytest.mark.parametrize("probability", [-0.1, 1.1, float("nan"), float("inf"), "0.8", True, None]) def test_typesafe_rejects_invalid_probabilities(probability: object) -> None: response = Mock(status_code=200) @@ -72,12 +183,11 @@ def test_typesafe_rejects_invalid_probabilities(probability: object) -> None: patch("products.ai_observability.backend.llm.system_one.pinned_request", return_value=response), pytest.raises(StructuredOutputParseError), ): - SystemOneClient.evaluate_boolean( + SystemOneClient.evaluate( api_key="test-typesafe-key", model="jev-1.13.0", - prompt="Is the response polite?", - source="Hello!", - allows_na=False, + state="Hello!", + questions={"verdict": NoulQuestion(instructions="Is the response polite?")}, ) @@ -88,12 +198,11 @@ def test_typesafe_rate_limits_are_retryable(status: int) -> None: patch("products.ai_observability.backend.llm.system_one.pinned_request", return_value=response), pytest.raises(SystemOneRateLimitError) as error, ): - SystemOneClient.evaluate_boolean( + SystemOneClient.evaluate( api_key="test-typesafe-key", model="jev-1.13.0", - prompt="Is the response polite?", - source="Hello!", - allows_na=False, + state="Hello!", + questions={"verdict": NoulQuestion(instructions="Is the response polite?")}, ) assert error.value.retry_after == 15 @@ -103,8 +212,8 @@ def test_typesafe_requires_a_key() -> None: patch("products.ai_observability.backend.llm.system_one.pinned_request") as request, pytest.raises(AuthenticationError), ): - SystemOneClient.evaluate_boolean( - api_key="", model="jev-1.13.0", prompt="Polite?", source="Hello!", allows_na=False + SystemOneClient.evaluate( + api_key="", model="jev-1.13.0", state="Hello!", questions={"verdict": NoulQuestion(instructions="Polite?")} ) request.assert_not_called() @@ -121,8 +230,14 @@ def test_typesafe_requires_every_requested_answer(answers: dict[str, object]) -> patch("products.ai_observability.backend.llm.system_one.pinned_request", return_value=response), pytest.raises(StructuredOutputParseError), ): - SystemOneClient.evaluate_boolean( - api_key="test-typesafe-key", model="jev-1.13.0", prompt="Polite?", source="Hello!", allows_na=True + SystemOneClient.evaluate( + api_key="test-typesafe-key", + model="jev-1.13.0", + state="Hello!", + questions={ + "verdict": NoulQuestion(instructions="Polite?"), + "applicable": NoulQuestion(instructions="Relevant?"), + }, ) @@ -143,8 +258,11 @@ def test_typesafe_preserves_error_categories(status: int, message: str, error_ty patch("products.ai_observability.backend.llm.system_one.pinned_request", return_value=response), pytest.raises(error_type), ): - SystemOneClient.evaluate_boolean( - api_key="test-typesafe-key", model="jev-1.13.0", prompt="Polite?", source="Hello!", allows_na=False + SystemOneClient.evaluate( + api_key="test-typesafe-key", + model="jev-1.13.0", + state="Hello!", + questions={"verdict": NoulQuestion(instructions="Polite?")}, ) @@ -158,19 +276,22 @@ def test_custom_endpoint_and_model(api_key: str) -> None: "latency_ms": 42, } with patch("products.ai_observability.backend.llm.system_one.pinned_request", return_value=response) as request: - result = SystemOneClient.evaluate_boolean( + result = SystemOneClient.evaluate( api_key=api_key, base_url="https://decisions.example.com/v1/", model="custom-model", - prompt="Polite?", - source="Hello!", - allows_na=True, + state="Hello!", + questions={ + "verdict": NoulQuestion(instructions="Polite?"), + "applicable": NoulQuestion(instructions="Relevant?"), + }, ) assert request.call_args.args == ("POST", "https://decisions.example.com/v1/systemone") assert request.call_args.kwargs["headers"] == ({"Authorization": f"Bearer {api_key}"} if api_key else {}) assert request.call_args.kwargs["json"]["model"] == "custom-model" assert result.model == "custom-model-revision" assert result.usage.output_tokens == 0 + assert isinstance(result.answers["verdict"], NoulAnswer) assert result.answers["verdict"].noul == 0.7 diff --git a/products/ai_observability/backend/models/evaluation_configs.py b/products/ai_observability/backend/models/evaluation_configs.py index a45fd06ec3f4..2c928a3eaf02 100644 --- a/products/ai_observability/backend/models/evaluation_configs.py +++ b/products/ai_observability/backend/models/evaluation_configs.py @@ -81,17 +81,35 @@ class NumericOutputConfig(BaseModel): min: float | None = None max: float | None = None step: float | None = Field(default=None, gt=0) + score_levels: list[str] | None = Field(default=None, min_length=2, max_length=10) allows_na: bool = False passing_rule: NumericPassingRule | None = None + @field_validator("score_levels") + @classmethod + def validate_score_levels(cls, value: list[str] | None) -> list[str] | None: + if value is not None: + value = [level.strip() for level in value] + if any(not level for level in value): + raise ValueError("Describe every score level") + return value + @model_validator(mode="after") def validate_bounds(self) -> Self: if self.min is not None and self.max is not None and self.min > self.max: raise ValueError("Minimum score cannot exceed maximum score") + if self.score_levels is not None and (self.min is None or self.max is None or self.min >= self.max): + raise ValueError("Score levels require a minimum and a greater maximum") if self.passing_rule is not None: self.passing_rule.threshold = self.validate_score(self.passing_rule.threshold) return self + def score_from_level(self, level: float) -> float: + if self.score_levels is None or self.min is None or self.max is None: + raise ValueError("Configure score levels and bounds before using this judge") + fraction = level / (len(self.score_levels) - 1) + return self.validate_score((1 - fraction) * self.min + fraction * self.max) + def validate_score(self, value: object) -> float: if isinstance(value, bool) or not isinstance(value, int | float): raise ValueError("Numeric evaluations must return a finite number") diff --git a/products/ai_observability/backend/models/provider_keys.py b/products/ai_observability/backend/models/provider_keys.py index f2077b51f7b1..1d3acbdc409e 100644 --- a/products/ai_observability/backend/models/provider_keys.py +++ b/products/ai_observability/backend/models/provider_keys.py @@ -22,7 +22,7 @@ class LLMProvider(models.TextChoices): TOGETHER_AI = "together_ai", "Together AI" MINIMAX = "minimax", "MiniMax" ZEABUR = "zeabur", "Zeabur AI Hub" - TYPESAFE = "typesafe", "System One (Jev)" + TYPESAFE = "typesafe", "System One" def llm_provider_choices() -> list[tuple[str, str | Promise]]: diff --git a/products/ai_observability/backend/models/test/test_evaluation_configs.py b/products/ai_observability/backend/models/test/test_evaluation_configs.py index f780826f31d5..f075adb85ec9 100644 --- a/products/ai_observability/backend/models/test/test_evaluation_configs.py +++ b/products/ai_observability/backend/models/test/test_evaluation_configs.py @@ -57,6 +57,12 @@ def test_numeric_configuration(self, runtime, config): {"passing_rule": {"operator": "gte", "threshold": True}}, {"passing_rule": {"operator": "gte", "threshold": float("nan")}}, {"min": 0, "passing_rule": {"operator": "gte", "threshold": -1}}, + {"score_levels": ["Poor", "Good"]}, + {"min": 0, "score_levels": ["Poor", "Good"]}, + {"min": 0, "max": 0, "score_levels": ["Poor", "Good"]}, + {"min": 0, "max": 10, "score_levels": ["Only one"]}, + {"min": 0, "max": 10, "score_levels": ["Level"] * 11}, + {"min": 0, "max": 10, "score_levels": ["Poor", " "]}, ], ) def test_invalid_numeric_configuration(self, output): @@ -72,6 +78,12 @@ def test_unbounded_and_nullable_configuration(self): ) assert output == {"allows_na": True} + @pytest.mark.parametrize("level,expected", [(0, -10), (1.25, 2.5), (2, 10)]) + def test_score_levels_keep_fractional_values(self, level: float, expected: float) -> None: + config = NumericOutputConfig(min=-10, max=10, score_levels=[" Poor ", "Fair", "Good"]) + assert config.score_levels == ["Poor", "Fair", "Good"] + assert config.score_from_level(level) == expected + class TestValidateTargetConfig: @pytest.mark.parametrize( diff --git a/products/ai_observability/frontend/components/EvaluationExplanation.tsx b/products/ai_observability/frontend/components/EvaluationExplanation.tsx index 1a6262e677d4..9cebc004cd3c 100644 --- a/products/ai_observability/frontend/components/EvaluationExplanation.tsx +++ b/products/ai_observability/frontend/components/EvaluationExplanation.tsx @@ -9,7 +9,7 @@ export function EvaluationExplanation({ }): JSX.Element { if (probability != null) { return ( - + {`${(probability * 100).toFixed(1)}% probability of true`} ) diff --git a/products/ai_observability/frontend/evaluations/AIObservabilityEvaluation.tsx b/products/ai_observability/frontend/evaluations/AIObservabilityEvaluation.tsx index 0b74aeeb660d..d32f6b3c1345 100644 --- a/products/ai_observability/frontend/evaluations/AIObservabilityEvaluation.tsx +++ b/products/ai_observability/frontend/evaluations/AIObservabilityEvaluation.tsx @@ -180,7 +180,10 @@ export function AIObservabilityEvaluation(): JSX.Element { : !hasSelectedJudgeModel ? 'Select a judge model before saving' : evaluation.output_type === 'numeric' - ? (numericOutputConfigError(evaluation.output_config) ?? undefined) + ? (numericOutputConfigError( + evaluation.output_config, + evaluation.model_configuration?.provider === 'typesafe' + ) ?? undefined) : undefined const focusTriggers = (): void => { @@ -746,6 +749,9 @@ export function AIObservabilityEvaluation(): JSX.Element { )} @@ -930,7 +936,10 @@ function EvaluationModelPicker(): JSX.Element { // Evals always run on the team's own provider key, so only BYOK models are offered. const selectedModelName = byokModels.find((m) => m.id === selectedModel)?.name const groups = evaluationProviderModelGroups.filter( - (group) => evaluation?.output_type === 'boolean' || group.provider !== 'typesafe' + (group) => + evaluation?.output_type === 'boolean' || + evaluation?.output_type === 'numeric' || + group.provider !== 'typesafe' ) const loading = byokModelsLoading || providerKeysLoading @@ -957,12 +966,13 @@ function EvaluationModelPicker(): JSX.Element { data-attr="evaluation-model-selector" /> - {evaluation?.model_configuration?.provider === 'typesafe' && ( -

- System One returns a probability without written reasoning. A probability of 50% or higher - produces a true result. -

- )} + {evaluation?.model_configuration?.provider === 'typesafe' && + evaluation.output_type === 'boolean' && ( +

+ This judge returns a probability without written reasoning. A probability of 50% or + higher produces a true result. +

+ )} {modelSelectionRequired && !selectedModel && (

Select a judge model.

)} diff --git a/products/ai_observability/frontend/evaluations/components/NumericEvaluationConfig.stories.tsx b/products/ai_observability/frontend/evaluations/components/NumericEvaluationConfig.stories.tsx index ccde3ae802c9..a34411314f2e 100644 --- a/products/ai_observability/frontend/evaluations/components/NumericEvaluationConfig.stories.tsx +++ b/products/ai_observability/frontend/evaluations/components/NumericEvaluationConfig.stories.tsx @@ -13,18 +13,22 @@ export default meta type Story = StoryObj -function Template({ narrow = false }: { narrow?: boolean }): JSX.Element { +function Template({ narrow = false, systemOne = false }: { narrow?: boolean; systemOne?: boolean }): JSX.Element { const [config, setConfig] = useState({ min: 0, max: 10, step: 0.5, allows_na: true, passing_rule: { operator: 'gte', threshold: 7 }, + ...(systemOne + ? { score_levels: ['Does not meet the criteria', 'Partly meets the criteria', 'Fully meets the criteria'] } + : {}), }) return (
setConfig((current) => ({ ...current, ...patch }))} />
@@ -33,3 +37,4 @@ function Template({ narrow = false }: { narrow?: boolean }): JSX.Element { export const Default: Story = { render: () =>