From 18a7f21cbefad6e7366e669a94203841e8913d5d Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 6 Oct 2026 14:15:36 +0000 Subject: [PATCH 01/25] chore!: require Python 3.11 or newer Python 3.10 reaches end of life in October 2026. Raise requires-python to >=3.11, target py311 in ruff, drop 3.10 from the CI unit matrix, and remove the pre-3.11 fallbacks for asyncio.create_task(context=...) and typing.NotRequired. Co-authored-by: Hassieb Pakzad --- .github/workflows/ci.yml | 1 - AGENTS.md | 2 +- langfuse/_client/observe.py | 29 +--- langfuse/types.py | 7 +- pyproject.toml | 4 +- tests/e2e/test_decorators.py | 4 - tests/unit/test_observe.py | 38 ----- uv.lock | 279 +---------------------------------- 8 files changed, 11 insertions(+), 353 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 8789b586e..104cda66f 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -70,7 +70,6 @@ jobs: fail-fast: false matrix: python-version: - - "3.10" - "3.11" - "3.12" - "3.13" diff --git a/AGENTS.md b/AGENTS.md index 1d82ae5dd..851b4e8d1 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -109,7 +109,7 @@ Minimum verification matrix: The main CI workflow currently runs: - linting on Python 3.13 - mypy on Python 3.13 -- `tests/unit` on a Python 3.10-3.14 matrix +- `tests/unit` on a Python 3.11-3.14 matrix - `tests/e2e` in 2 mechanical shards plus a serial subset inside each shard - `tests/live_provider` as one always-on suite - PR title validation for Conventional Commits diff --git a/langfuse/_client/observe.py b/langfuse/_client/observe.py index 848506a46..9d28d8156 100644 --- a/langfuse/_client/observe.py +++ b/langfuse/_client/observe.py @@ -2,7 +2,6 @@ import contextvars import inspect import os -import sys from functools import wraps from typing import ( Any, @@ -49,8 +48,6 @@ P = ParamSpec("P") R = TypeVar("R") -_ASYNCIO_CREATE_TASK_SUPPORTS_CONTEXT = sys.version_info >= (3, 11) - class LangfuseDecorator: """Implementation of the @observe decorator for seamless Langfuse tracing integration. @@ -217,7 +214,7 @@ def decorator(func: F) -> F: capture_output=should_capture_output, transform_to_string=transform_to_string, ) - if asyncio.iscoroutinefunction(func) + if inspect.iscoroutinefunction(func) else self._sync_observe( func, name=name, @@ -747,15 +744,7 @@ async def aclose(self) -> None: self._finalize() async def _close_generator(self) -> None: - if _ASYNCIO_CREATE_TASK_SUPPORTS_CONTEXT: - close_task = asyncio.create_task( - self.generator.aclose(), - context=self.context, - ) # type: ignore - else: - close_task = self.context.run(asyncio.create_task, self.generator.aclose()) - - await close_task + await asyncio.create_task(self.generator.aclose(), context=self.context) async def close(self) -> None: await self.aclose() @@ -769,16 +758,10 @@ def __del__(self) -> None: async def __anext__(self) -> Any: try: # Run the generator's __anext__ in the preserved context - if _ASYNCIO_CREATE_TASK_SUPPORTS_CONTEXT: - item = await asyncio.create_task( - self.generator.__anext__(), # type: ignore - context=self.context, - ) # type: ignore - else: - item = await self.context.run( - asyncio.create_task, - self.generator.__anext__(), # type: ignore - ) + item = await asyncio.create_task( + self.generator.__anext__(), # type: ignore + context=self.context, + ) if self.capture_output: self.items.append(item) diff --git a/langfuse/types.py b/langfuse/types.py index 66bddbca6..d20bd6a55 100644 --- a/langfuse/types.py +++ b/langfuse/types.py @@ -24,6 +24,7 @@ def my_evaluator(*, output: str, **kwargs) -> Evaluation: Dict, Literal, Mapping, + NotRequired, Optional, Protocol, Sequence, @@ -31,12 +32,6 @@ def my_evaluator(*, output: str, **kwargs) -> Evaluation: Union, ) -try: - from typing import NotRequired # type: ignore -except ImportError: - from typing_extensions import NotRequired - - from langfuse.api import MediaContentType # Span attribute values accepted by the OpenTelemetry trace API. OpenTelemetry 1.45 diff --git a/pyproject.toml b/pyproject.toml index 60e73a9ec..1534e4aa6 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -6,7 +6,7 @@ readme = "README.md" authors = [{ name = "langfuse", email = "developers@langfuse.com" }] license = "MIT" license-files = ["LICENSE"] -requires-python = ">=3.10,<4.0" +requires-python = ">=3.11,<4.0" keywords = [ "langfuse", "llm", @@ -139,7 +139,7 @@ module = "langfuse.api.*" disable_error_code = ["redundant-cast", "no-untyped-def"] [tool.ruff] -target-version = "py310" +target-version = "py311" [tool.ruff.lint] extend-select = [ diff --git a/tests/e2e/test_decorators.py b/tests/e2e/test_decorators.py index 9e302f799..9fac4cf22 100644 --- a/tests/e2e/test_decorators.py +++ b/tests/e2e/test_decorators.py @@ -1,6 +1,5 @@ import asyncio import os -import sys from collections import defaultdict from concurrent.futures import ThreadPoolExecutor from time import sleep @@ -1880,7 +1879,6 @@ def root_function(): @pytest.mark.asyncio -@pytest.mark.skipif(sys.version_info < (3, 11), reason="requires python3.11 or higher") async def test_async_generator_context_preservation(): """Test that async generators preserve context when consumed later (e.g., by streaming responses)""" langfuse = get_client() @@ -1946,7 +1944,6 @@ async def root_function(): @pytest.mark.asyncio -@pytest.mark.skipif(sys.version_info < (3, 11), reason="requires python3.11 or higher") async def test_async_generator_context_preservation_with_trace_hierarchy(): """Test that async generators maintain proper parent-child span relationships""" langfuse = get_client() @@ -2009,7 +2006,6 @@ async def parent_function(): @pytest.mark.asyncio -@pytest.mark.skipif(sys.version_info < (3, 11), reason="requires python3.11 or higher") async def test_async_generator_exception_handling_with_context(): """Test that exceptions in async generators are properly handled while preserving context""" langfuse = get_client() diff --git a/tests/unit/test_observe.py b/tests/unit/test_observe.py index f2ff11789..865779591 100644 --- a/tests/unit/test_observe.py +++ b/tests/unit/test_observe.py @@ -3,13 +3,11 @@ import gc import inspect import json -import sys from typing import Any, AsyncGenerator, Generator, cast import pytest from langfuse import observe -from langfuse._client import observe as observe_module from langfuse._client.attributes import LangfuseOtelSpanAttributes from langfuse._client.observe import ( _ContextPreservedAsyncGeneratorWrapper, @@ -96,7 +94,6 @@ def body() -> Generator[str, None, None]: @pytest.mark.asyncio -@pytest.mark.skipif(sys.version_info < (3, 11), reason="requires python3.11 or higher") async def test_streaming_response_preserves_context_without_output_capture( langfuse_memory_client: Any, memory_exporter: Any ) -> None: @@ -479,41 +476,6 @@ async def generator() -> AsyncGenerator[str, None]: assert span.updates[-1] == {"level": "ERROR", "status_message": "cleanup failed"} -@pytest.mark.asyncio -async def test_async_generator_wrapper_fallback_preserves_context( - monkeypatch: pytest.MonkeyPatch, -) -> None: - marker = contextvars.ContextVar("marker", default="ambient") - seen: list[str] = [] - monkeypatch.setattr(observe_module, "_ASYNCIO_CREATE_TASK_SUPPORTS_CONTEXT", False) - - async def generator() -> AsyncGenerator[str, None]: - try: - yield marker.get() - yield "item_1" - finally: - seen.append(marker.get()) - - span = SpanRecorder() - context = contextvars.copy_context() - context.run(marker.set, "preserved") - wrapper = _ContextPreservedAsyncGeneratorWrapper( - generator(), - context, - cast(Any, span), - False, - None, - ) - - assert await wrapper.__anext__() == "preserved" - marker.set("ambient-now") - - await wrapper.aclose() - - assert seen == ["preserved"] - assert span.ended == 1 - - @pytest.mark.asyncio async def test_async_generator_wrapper_del_ends_span_when_abandoned() -> None: async def generator() -> AsyncGenerator[str, None]: diff --git a/uv.lock b/uv.lock index a71f89727..ecacc3a0d 100644 --- a/uv.lock +++ b/uv.lock @@ -1,6 +1,6 @@ version = 1 revision = 3 -requires-python = ">=3.10, <4.0" +requires-python = ">=3.11, <4.0" [options] exclude-newer = "2026-09-28T08:53:26.348016113Z" @@ -32,7 +32,6 @@ name = "anyio" version = "4.12.1" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "exceptiongroup", marker = "python_full_version < '3.11'" }, { name = "idna" }, { name = "typing-extensions", marker = "python_full_version < '3.13'" }, ] @@ -74,15 +73,6 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/df/73/b6e24bd22e6720ca8ee9a85a0c4a2971af8497d8f3193fa05390cbd46e09/backoff-2.2.1-py3-none-any.whl", hash = "sha256:63579f9a0628e06278f7e47b7d7d5b6ce20dc65c5e96a6f3ca99a6adca0396e8", size = 15148, upload-time = "2022-10-05T19:19:30.546Z" }, ] -[[package]] -name = "backports-asyncio-runner" -version = "1.2.0" -source = { registry = "https://pypi.org/simple" } -sdist = { url = "https://files.pythonhosted.org/packages/8e/ff/70dca7d7cb1cbc0edb2c6cc0c38b65cba36cccc491eca64cabd5fe7f8670/backports_asyncio_runner-1.2.0.tar.gz", hash = "sha256:a5aa7b2b7d8f8bfcaa2b57313f70792df84e32a2a746f585213373f900b42162", size = 69893, upload-time = "2025-07-02T02:27:15.685Z" } -wheels = [ - { url = "https://files.pythonhosted.org/packages/a0/59/76ab57e3fe74484f48a53f8e337171b4a2349e506eabe136d7e01d059086/backports_asyncio_runner-1.2.0-py3-none-any.whl", hash = "sha256:0da0a936a8aeb554eccb426dc55af3ba63bcdc69fa1a600b5bb305413a4477b5", size = 12313, upload-time = "2025-07-02T02:27:14.263Z" }, -] - [[package]] name = "certifi" version = "2026.2.25" @@ -107,22 +97,6 @@ version = "3.4.6" source = { registry = "https://pypi.org/simple" } sdist = { url = "https://files.pythonhosted.org/packages/7b/60/e3bec1881450851b087e301bedc3daa9377a4d45f1c26aa90b0b235e38aa/charset_normalizer-3.4.6.tar.gz", hash = "sha256:1ae6b62897110aa7c79ea2f5dd38d1abca6db663687c0b1ad9aed6f6bae3d9d6", size = 143363, upload-time = "2026-03-15T18:53:25.478Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/e6/8c/2c56124c6dc53a774d435f985b5973bc592f42d437be58c0c92d65ae7296/charset_normalizer-3.4.6-cp310-cp310-macosx_10_9_universal2.whl", hash = "sha256:2e1d8ca8611099001949d1cdfaefc510cf0f212484fe7c565f735b68c78c3c95", size = 298751, upload-time = "2026-03-15T18:50:00.003Z" }, - { url = "https://files.pythonhosted.org/packages/86/2a/2a7db6b314b966a3bcad8c731c0719c60b931b931de7ae9f34b2839289ee/charset_normalizer-3.4.6-cp310-cp310-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:e25369dc110d58ddf29b949377a93e0716d72a24f62bad72b2b39f155949c1fd", size = 200027, upload-time = "2026-03-15T18:50:01.702Z" }, - { url = "https://files.pythonhosted.org/packages/68/f2/0fe775c74ae25e2a3b07b01538fc162737b3e3f795bada3bc26f4d4d495c/charset_normalizer-3.4.6-cp310-cp310-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:259695e2ccc253feb2a016303543d691825e920917e31f894ca1a687982b1de4", size = 220741, upload-time = "2026-03-15T18:50:03.194Z" }, - { url = "https://files.pythonhosted.org/packages/10/98/8085596e41f00b27dd6aa1e68413d1ddda7e605f34dd546833c61fddd709/charset_normalizer-3.4.6-cp310-cp310-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:dda86aba335c902b6149a02a55b38e96287157e609200811837678214ba2b1db", size = 215802, upload-time = "2026-03-15T18:50:05.859Z" }, - { url = "https://files.pythonhosted.org/packages/fd/ce/865e4e09b041bad659d682bbd98b47fb490b8e124f9398c9448065f64fee/charset_normalizer-3.4.6-cp310-cp310-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:51fb3c322c81d20567019778cb5a4a6f2dc1c200b886bc0d636238e364848c89", size = 207908, upload-time = "2026-03-15T18:50:07.676Z" }, - { url = "https://files.pythonhosted.org/packages/a8/54/8c757f1f7349262898c2f169e0d562b39dcb977503f18fdf0814e923db78/charset_normalizer-3.4.6-cp310-cp310-manylinux_2_31_armv7l.whl", hash = "sha256:4482481cb0572180b6fd976a4d5c72a30263e98564da68b86ec91f0fe35e8565", size = 194357, upload-time = "2026-03-15T18:50:09.327Z" }, - { url = "https://files.pythonhosted.org/packages/6f/29/e88f2fac9218907fc7a70722b393d1bbe8334c61fe9c46640dba349b6e66/charset_normalizer-3.4.6-cp310-cp310-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:39f5068d35621da2881271e5c3205125cc456f54e9030d3f723288c873a71bf9", size = 205610, upload-time = "2026-03-15T18:50:10.732Z" }, - { url = "https://files.pythonhosted.org/packages/4c/c5/21d7bb0cb415287178450171d130bed9d664211fdd59731ed2c34267b07d/charset_normalizer-3.4.6-cp310-cp310-musllinux_1_2_aarch64.whl", hash = "sha256:8bea55c4eef25b0b19a0337dc4e3f9a15b00d569c77211fa8cde38684f234fb7", size = 203512, upload-time = "2026-03-15T18:50:12.535Z" }, - { url = "https://files.pythonhosted.org/packages/a4/be/ce52f3c7fdb35cc987ad38a53ebcef52eec498f4fb6c66ecfe62cfe57ba2/charset_normalizer-3.4.6-cp310-cp310-musllinux_1_2_armv7l.whl", hash = "sha256:f0cdaecd4c953bfae0b6bb64910aaaca5a424ad9c72d85cb88417bb9814f7550", size = 195398, upload-time = "2026-03-15T18:50:14.236Z" }, - { url = "https://files.pythonhosted.org/packages/81/a0/3ab5dd39d4859a3555e5dadfc8a9fa7f8352f8c183d1a65c90264517da0e/charset_normalizer-3.4.6-cp310-cp310-musllinux_1_2_ppc64le.whl", hash = "sha256:150b8ce8e830eb7ccb029ec9ca36022f756986aaaa7956aad6d9ec90089338c0", size = 221772, upload-time = "2026-03-15T18:50:15.581Z" }, - { url = "https://files.pythonhosted.org/packages/04/6e/6a4e41a97ba6b2fa87f849c41e4d229449a586be85053c4d90135fe82d26/charset_normalizer-3.4.6-cp310-cp310-musllinux_1_2_riscv64.whl", hash = "sha256:e68c14b04827dd76dcbd1aeea9e604e3e4b78322d8faf2f8132c7138efa340a8", size = 205759, upload-time = "2026-03-15T18:50:17.047Z" }, - { url = "https://files.pythonhosted.org/packages/db/3b/34a712a5ee64a6957bf355b01dc17b12de457638d436fdb05d01e463cd1c/charset_normalizer-3.4.6-cp310-cp310-musllinux_1_2_s390x.whl", hash = "sha256:3778fd7d7cd04ae8f54651f4a7a0bd6e39a0cf20f801720a4c21d80e9b7ad6b0", size = 216938, upload-time = "2026-03-15T18:50:18.44Z" }, - { url = "https://files.pythonhosted.org/packages/cb/05/5bd1e12da9ab18790af05c61aafd01a60f489778179b621ac2a305243c62/charset_normalizer-3.4.6-cp310-cp310-musllinux_1_2_x86_64.whl", hash = "sha256:dad6e0f2e481fffdcf776d10ebee25e0ef89f16d691f1e5dee4b586375fdc64b", size = 210138, upload-time = "2026-03-15T18:50:19.852Z" }, - { url = "https://files.pythonhosted.org/packages/bd/8e/3cb9e2d998ff6b21c0a1860343cb7b83eba9cdb66b91410e18fc4969d6ab/charset_normalizer-3.4.6-cp310-cp310-win32.whl", hash = "sha256:74a2e659c7ecbc73562e2a15e05039f1e22c75b7c7618b4b574a3ea9118d1557", size = 144137, upload-time = "2026-03-15T18:50:21.505Z" }, - { url = "https://files.pythonhosted.org/packages/d8/8f/78f5489ffadb0db3eb7aff53d31c24531d33eb545f0c6f6567c25f49a5ff/charset_normalizer-3.4.6-cp310-cp310-win_amd64.whl", hash = "sha256:aa9cccf4a44b9b62d8ba8b4dd06c649ba683e4bf04eea606d2e94cfc2d6ff4d6", size = 154244, upload-time = "2026-03-15T18:50:22.81Z" }, - { url = "https://files.pythonhosted.org/packages/e4/74/e472659dffb0cadb2f411282d2d76c60da1fc94076d7fffed4ae8a93ec01/charset_normalizer-3.4.6-cp310-cp310-win_arm64.whl", hash = "sha256:e985a16ff513596f217cee86c21371b8cd011c0f6f056d0920aa2d926c544058", size = 143312, upload-time = "2026-03-15T18:50:24.074Z" }, { url = "https://files.pythonhosted.org/packages/62/28/ff6f234e628a2de61c458be2779cb182bc03f6eec12200d4a525bbfc9741/charset_normalizer-3.4.6-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:82060f995ab5003a2d6e0f4ad29065b7672b6593c8c63559beefe5b443242c3e", size = 293582, upload-time = "2026-03-15T18:50:25.454Z" }, { url = "https://files.pythonhosted.org/packages/1c/b7/b1a117e5385cbdb3205f6055403c2a2a220c5ea80b8716c324eaf75c5c95/charset_normalizer-3.4.6-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:60c74963d8350241a79cb8feea80e54d518f72c26db618862a8f53e5023deaf9", size = 197240, upload-time = "2026-03-15T18:50:27.196Z" }, { url = "https://files.pythonhosted.org/packages/a1/5f/2574f0f09f3c3bc1b2f992e20bce6546cb1f17e111c5be07308dc5427956/charset_normalizer-3.4.6-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:f6e4333fb15c83f7d1482a76d45a0818897b3d33f00efd215528ff7c51b8e35d", size = 217363, upload-time = "2026-03-15T18:50:28.601Z" }, @@ -242,18 +216,6 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/12/b3/231ffd4ab1fc9d679809f356cebee130ac7daa00d6d6f3206dd4fd137e9e/distro-1.9.0-py3-none-any.whl", hash = "sha256:7bffd925d65168f85027d8da9af6bddab658135b840670a223589bc0c8ef02b2", size = 20277, upload-time = "2023-12-24T09:54:30.421Z" }, ] -[[package]] -name = "exceptiongroup" -version = "1.3.1" -source = { registry = "https://pypi.org/simple" } -dependencies = [ - { name = "typing-extensions", marker = "python_full_version < '3.13'" }, -] -sdist = { url = "https://files.pythonhosted.org/packages/50/79/66800aadf48771f6b62f7eb014e352e5d06856655206165d775e675a02c9/exceptiongroup-1.3.1.tar.gz", hash = "sha256:8b412432c6055b0b7d14c310000ae93352ed6754f70fa8f7c34141f91c4e3219", size = 30371, upload-time = "2025-11-21T23:01:54.787Z" } -wheels = [ - { url = "https://files.pythonhosted.org/packages/8a/0e/97c33bf5009bdbac74fd2beace167cab3f978feb69cc36f1ef79360d6c4e/exceptiongroup-1.3.1-py3-none-any.whl", hash = "sha256:a7a39a3bd276781e98394987d3a5701d0c4edffb633bb7a5144577f82c773598", size = 16740, upload-time = "2025-11-21T23:01:53.443Z" }, -] - [[package]] name = "execnet" version = "2.1.2" @@ -366,18 +328,6 @@ version = "0.13.0" source = { registry = "https://pypi.org/simple" } sdist = { url = "https://files.pythonhosted.org/packages/0d/5e/4ec91646aee381d01cdb9974e30882c9cd3b8c5d1079d6b5ff4af522439a/jiter-0.13.0.tar.gz", hash = "sha256:f2839f9c2c7e2dffc1bc5929a510e14ce0a946be9365fd1219e7ef342dae14f4", size = 164847, upload-time = "2026-02-02T12:37:56.441Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/d0/5a/41da76c5ea07bec1b0472b6b2fdb1b651074d504b19374d7e130e0cdfb25/jiter-0.13.0-cp310-cp310-macosx_10_12_x86_64.whl", hash = "sha256:2ffc63785fd6c7977defe49b9824ae6ce2b2e2b77ce539bdaf006c26da06342e", size = 311164, upload-time = "2026-02-02T12:35:17.688Z" }, - { url = "https://files.pythonhosted.org/packages/40/cb/4a1bf994a3e869f0d39d10e11efb471b76d0ad70ecbfb591427a46c880c2/jiter-0.13.0-cp310-cp310-macosx_11_0_arm64.whl", hash = "sha256:4a638816427006c1e3f0013eb66d391d7a3acda99a7b0cf091eff4497ccea33a", size = 320296, upload-time = "2026-02-02T12:35:19.828Z" }, - { url = "https://files.pythonhosted.org/packages/09/82/acd71ca9b50ecebadc3979c541cd717cce2fe2bc86236f4fa597565d8f1a/jiter-0.13.0-cp310-cp310-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:19928b5d1ce0ff8c1ee1b9bdef3b5bfc19e8304f1b904e436caf30bc15dc6cf5", size = 352742, upload-time = "2026-02-02T12:35:21.258Z" }, - { url = "https://files.pythonhosted.org/packages/71/03/d1fc996f3aecfd42eb70922edecfb6dd26421c874503e241153ad41df94f/jiter-0.13.0-cp310-cp310-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:309549b778b949d731a2f0e1594a3f805716be704a73bf3ad9a807eed5eb5721", size = 363145, upload-time = "2026-02-02T12:35:24.653Z" }, - { url = "https://files.pythonhosted.org/packages/f1/61/a30492366378cc7a93088858f8991acd7d959759fe6138c12a4644e58e81/jiter-0.13.0-cp310-cp310-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:bcdabaea26cb04e25df3103ce47f97466627999260290349a88c8136ecae0060", size = 487683, upload-time = "2026-02-02T12:35:26.162Z" }, - { url = "https://files.pythonhosted.org/packages/20/4e/4223cffa9dbbbc96ed821c5aeb6bca510848c72c02086d1ed3f1da3d58a7/jiter-0.13.0-cp310-cp310-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:a3a377af27b236abbf665a69b2bdd680e3b5a0bd2af825cd3b81245279a7606c", size = 373579, upload-time = "2026-02-02T12:35:27.582Z" }, - { url = "https://files.pythonhosted.org/packages/fe/c9/b0489a01329ab07a83812d9ebcffe7820a38163c6d9e7da644f926ff877c/jiter-0.13.0-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:fe49d3ff6db74321f144dff9addd4a5874d3105ac5ba7c5b77fac099cfae31ae", size = 362904, upload-time = "2026-02-02T12:35:28.925Z" }, - { url = "https://files.pythonhosted.org/packages/05/af/53e561352a44afcba9a9bc67ee1d320b05a370aed8df54eafe714c4e454d/jiter-0.13.0-cp310-cp310-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:2113c17c9a67071b0f820733c0893ed1d467b5fcf4414068169e5c2cabddb1e2", size = 392380, upload-time = "2026-02-02T12:35:30.385Z" }, - { url = "https://files.pythonhosted.org/packages/76/2a/dd805c3afb8ed5b326c5ae49e725d1b1255b9754b1b77dbecdc621b20773/jiter-0.13.0-cp310-cp310-musllinux_1_1_aarch64.whl", hash = "sha256:ab1185ca5c8b9491b55ebf6c1e8866b8f68258612899693e24a92c5fdb9455d5", size = 517939, upload-time = "2026-02-02T12:35:31.865Z" }, - { url = "https://files.pythonhosted.org/packages/20/2a/7b67d76f55b8fe14c937e7640389612f05f9a4145fc28ae128aaa5e62257/jiter-0.13.0-cp310-cp310-musllinux_1_1_x86_64.whl", hash = "sha256:9621ca242547edc16400981ca3231e0c91c0c4c1ab8573a596cd9bb3575d5c2b", size = 551696, upload-time = "2026-02-02T12:35:33.306Z" }, - { url = "https://files.pythonhosted.org/packages/85/9c/57cdd64dac8f4c6ab8f994fe0eb04dc9fd1db102856a4458fcf8a99dfa62/jiter-0.13.0-cp310-cp310-win32.whl", hash = "sha256:a7637d92b1c9d7a771e8c56f445c7f84396d48f2e756e5978840ecba2fac0894", size = 204592, upload-time = "2026-02-02T12:35:34.58Z" }, - { url = "https://files.pythonhosted.org/packages/a7/38/f4f3ea5788b8a5bae7510a678cdc747eda0c45ffe534f9878ff37e7cf3b3/jiter-0.13.0-cp310-cp310-win_amd64.whl", hash = "sha256:c1b609e5cbd2f52bb74fb721515745b407df26d7b800458bd97cb3b972c29e7d", size = 206016, upload-time = "2026-02-02T12:35:36.435Z" }, { url = "https://files.pythonhosted.org/packages/71/29/499f8c9eaa8a16751b1c0e45e6f5f1761d180da873d417996cc7bddc8eef/jiter-0.13.0-cp311-cp311-macosx_10_12_x86_64.whl", hash = "sha256:ea026e70a9a28ebbdddcbcf0f1323128a8db66898a06eaad3a4e62d2f554d096", size = 311157, upload-time = "2026-02-02T12:35:37.758Z" }, { url = "https://files.pythonhosted.org/packages/50/f6/566364c777d2ab450b92100bea11333c64c38d32caf8dc378b48e5b20c46/jiter-0.13.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:66aa3e663840152d18cc8ff1e4faad3dd181373491b9cfdc6004b92198d67911", size = 319729, upload-time = "2026-02-02T12:35:39.246Z" }, { url = "https://files.pythonhosted.org/packages/73/dd/560f13ec5e4f116d8ad2658781646cca91b617ae3b8758d4a5076b278f70/jiter-0.13.0-cp311-cp311-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:c3524798e70655ff19aec58c7d05adb1f074fecff62da857ea9be2b908b6d701", size = 354766, upload-time = "2026-02-02T12:35:40.662Z" }, @@ -705,18 +655,6 @@ version = "0.8.1" source = { registry = "https://pypi.org/simple" } sdist = { url = "https://files.pythonhosted.org/packages/56/9c/b4b0c54d84da4a94b37bd44151e46d5e583c9534c7e02250b961b1b6d8a8/librt-0.8.1.tar.gz", hash = "sha256:be46a14693955b3bd96014ccbdb8339ee8c9346fbe11c1b78901b55125f14c73", size = 177471, upload-time = "2026-02-17T16:13:06.101Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/7c/5f/63f5fa395c7a8a93558c0904ba8f1c8d1b997ca6a3de61bc7659970d66bf/librt-0.8.1-cp310-cp310-macosx_10_9_x86_64.whl", hash = "sha256:81fd938344fecb9373ba1b155968c8a329491d2ce38e7ddb76f30ffb938f12dc", size = 65697, upload-time = "2026-02-17T16:11:06.903Z" }, - { url = "https://files.pythonhosted.org/packages/ff/e0/0472cf37267b5920eff2f292ccfaede1886288ce35b7f3203d8de00abfe6/librt-0.8.1-cp310-cp310-macosx_11_0_arm64.whl", hash = "sha256:5db05697c82b3a2ec53f6e72b2ed373132b0c2e05135f0696784e97d7f5d48e7", size = 68376, upload-time = "2026-02-17T16:11:08.395Z" }, - { url = "https://files.pythonhosted.org/packages/c8/be/8bd1359fdcd27ab897cd5963294fa4a7c83b20a8564678e4fd12157e56a5/librt-0.8.1-cp310-cp310-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", hash = "sha256:d56bc4011975f7460bea7b33e1ff425d2f1adf419935ff6707273c77f8a4ada6", size = 197084, upload-time = "2026-02-17T16:11:09.774Z" }, - { url = "https://files.pythonhosted.org/packages/e2/fe/163e33fdd091d0c2b102f8a60cc0a61fd730ad44e32617cd161e7cd67a01/librt-0.8.1-cp310-cp310-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:5cdc0f588ff4b663ea96c26d2a230c525c6fc62b28314edaaaca8ed5af931ad0", size = 207337, upload-time = "2026-02-17T16:11:11.311Z" }, - { url = "https://files.pythonhosted.org/packages/01/99/f85130582f05dcf0c8902f3d629270231d2f4afdfc567f8305a952ac7f14/librt-0.8.1-cp310-cp310-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:97c2b54ff6717a7a563b72627990bec60d8029df17df423f0ed37d56a17a176b", size = 219980, upload-time = "2026-02-17T16:11:12.499Z" }, - { url = "https://files.pythonhosted.org/packages/6f/54/cb5e4d03659e043a26c74e08206412ac9a3742f0477d96f9761a55313b5f/librt-0.8.1-cp310-cp310-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:8f1125e6bbf2f1657d9a2f3ccc4a2c9b0c8b176965bb565dd4d86be67eddb4b6", size = 212921, upload-time = "2026-02-17T16:11:14.484Z" }, - { url = "https://files.pythonhosted.org/packages/b1/81/a3a01e4240579c30f3487f6fed01eb4bc8ef0616da5b4ebac27ca19775f3/librt-0.8.1-cp310-cp310-musllinux_1_2_aarch64.whl", hash = "sha256:8f4bb453f408137d7581be309b2fbc6868a80e7ef60c88e689078ee3a296ae71", size = 221381, upload-time = "2026-02-17T16:11:17.459Z" }, - { url = "https://files.pythonhosted.org/packages/08/b0/fc2d54b4b1c6fb81e77288ff31ff25a2c1e62eaef4424a984f228839717b/librt-0.8.1-cp310-cp310-musllinux_1_2_i686.whl", hash = "sha256:c336d61d2fe74a3195edc1646d53ff1cddd3a9600b09fa6ab75e5514ba4862a7", size = 216714, upload-time = "2026-02-17T16:11:19.197Z" }, - { url = "https://files.pythonhosted.org/packages/96/96/85daa73ffbd87e1fb287d7af6553ada66bf25a2a6b0de4764344a05469f6/librt-0.8.1-cp310-cp310-musllinux_1_2_riscv64.whl", hash = "sha256:eb5656019db7c4deacf0c1a55a898c5bb8f989be904597fcb5232a2f4828fa05", size = 214777, upload-time = "2026-02-17T16:11:20.443Z" }, - { url = "https://files.pythonhosted.org/packages/12/9c/c3aa7a2360383f4bf4f04d98195f2739a579128720c603f4807f006a4225/librt-0.8.1-cp310-cp310-musllinux_1_2_x86_64.whl", hash = "sha256:c25d9e338d5bed46c1632f851babf3d13c78f49a225462017cf5e11e845c5891", size = 237398, upload-time = "2026-02-17T16:11:22.083Z" }, - { url = "https://files.pythonhosted.org/packages/61/19/d350ea89e5274665185dabc4bbb9c3536c3411f862881d316c8b8e00eb66/librt-0.8.1-cp310-cp310-win32.whl", hash = "sha256:aaab0e307e344cb28d800957ef3ec16605146ef0e59e059a60a176d19543d1b7", size = 54285, upload-time = "2026-02-17T16:11:23.27Z" }, - { url = "https://files.pythonhosted.org/packages/4f/d6/45d587d3d41c112e9543a0093d883eb57a24a03e41561c127818aa2a6bcc/librt-0.8.1-cp310-cp310-win_amd64.whl", hash = "sha256:56e04c14b696300d47b3bc5f1d10a00e86ae978886d0cee14e5714fafb5df5d2", size = 61352, upload-time = "2026-02-17T16:11:24.207Z" }, { url = "https://files.pythonhosted.org/packages/1d/01/0e748af5e4fee180cf7cd12bd12b0513ad23b045dccb2a83191bde82d168/librt-0.8.1-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:681dc2451d6d846794a828c16c22dc452d924e9f700a485b7ecb887a30aad1fd", size = 65315, upload-time = "2026-02-17T16:11:25.152Z" }, { url = "https://files.pythonhosted.org/packages/9d/4d/7184806efda571887c798d573ca4134c80ac8642dcdd32f12c31b939c595/librt-0.8.1-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:a3b4350b13cc0e6f5bec8fa7caf29a8fb8cdc051a3bae45cfbfd7ce64f009965", size = 68021, upload-time = "2026-02-17T16:11:26.129Z" }, { url = "https://files.pythonhosted.org/packages/ae/88/c3c52d2a5d5101f28d3dc89298444626e7874aa904eed498464c2af17627/librt-0.8.1-cp311-cp311-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", hash = "sha256:ac1e7817fd0ed3d14fd7c5df91daed84c48e4c2a11ee99c0547f9f62fdae13da", size = 194500, upload-time = "2026-02-17T16:11:27.177Z" }, @@ -790,17 +728,6 @@ version = "3.0.3" source = { registry = "https://pypi.org/simple" } sdist = { url = "https://files.pythonhosted.org/packages/7e/99/7690b6d4034fffd95959cbe0c02de8deb3098cc577c67bb6a24fe5d7caa7/markupsafe-3.0.3.tar.gz", hash = "sha256:722695808f4b6457b320fdc131280796bdceb04ab50fe1795cd540799ebe1698", size = 80313, upload-time = "2025-09-27T18:37:40.426Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/e8/4b/3541d44f3937ba468b75da9eebcae497dcf67adb65caa16760b0a6807ebb/markupsafe-3.0.3-cp310-cp310-macosx_10_9_x86_64.whl", hash = "sha256:2f981d352f04553a7171b8e44369f2af4055f888dfb147d55e42d29e29e74559", size = 11631, upload-time = "2025-09-27T18:36:05.558Z" }, - { url = "https://files.pythonhosted.org/packages/98/1b/fbd8eed11021cabd9226c37342fa6ca4e8a98d8188a8d9b66740494960e4/markupsafe-3.0.3-cp310-cp310-macosx_11_0_arm64.whl", hash = "sha256:e1c1493fb6e50ab01d20a22826e57520f1284df32f2d8601fdd90b6304601419", size = 12057, upload-time = "2025-09-27T18:36:07.165Z" }, - { url = "https://files.pythonhosted.org/packages/40/01/e560d658dc0bb8ab762670ece35281dec7b6c1b33f5fbc09ebb57a185519/markupsafe-3.0.3-cp310-cp310-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:1ba88449deb3de88bd40044603fafffb7bc2b055d626a330323a9ed736661695", size = 22050, upload-time = "2025-09-27T18:36:08.005Z" }, - { url = "https://files.pythonhosted.org/packages/af/cd/ce6e848bbf2c32314c9b237839119c5a564a59725b53157c856e90937b7a/markupsafe-3.0.3-cp310-cp310-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:f42d0984e947b8adf7dd6dde396e720934d12c506ce84eea8476409563607591", size = 20681, upload-time = "2025-09-27T18:36:08.881Z" }, - { url = "https://files.pythonhosted.org/packages/c9/2a/b5c12c809f1c3045c4d580b035a743d12fcde53cf685dbc44660826308da/markupsafe-3.0.3-cp310-cp310-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:c0c0b3ade1c0b13b936d7970b1d37a57acde9199dc2aecc4c336773e1d86049c", size = 20705, upload-time = "2025-09-27T18:36:10.131Z" }, - { url = "https://files.pythonhosted.org/packages/cf/e3/9427a68c82728d0a88c50f890d0fc072a1484de2f3ac1ad0bfc1a7214fd5/markupsafe-3.0.3-cp310-cp310-musllinux_1_2_aarch64.whl", hash = "sha256:0303439a41979d9e74d18ff5e2dd8c43ed6c6001fd40e5bf2e43f7bd9bbc523f", size = 21524, upload-time = "2025-09-27T18:36:11.324Z" }, - { url = "https://files.pythonhosted.org/packages/bc/36/23578f29e9e582a4d0278e009b38081dbe363c5e7165113fad546918a232/markupsafe-3.0.3-cp310-cp310-musllinux_1_2_riscv64.whl", hash = "sha256:d2ee202e79d8ed691ceebae8e0486bd9a2cd4794cec4824e1c99b6f5009502f6", size = 20282, upload-time = "2025-09-27T18:36:12.573Z" }, - { url = "https://files.pythonhosted.org/packages/56/21/dca11354e756ebd03e036bd8ad58d6d7168c80ce1fe5e75218e4945cbab7/markupsafe-3.0.3-cp310-cp310-musllinux_1_2_x86_64.whl", hash = "sha256:177b5253b2834fe3678cb4a5f0059808258584c559193998be2601324fdeafb1", size = 20745, upload-time = "2025-09-27T18:36:13.504Z" }, - { url = "https://files.pythonhosted.org/packages/87/99/faba9369a7ad6e4d10b6a5fbf71fa2a188fe4a593b15f0963b73859a1bbd/markupsafe-3.0.3-cp310-cp310-win32.whl", hash = "sha256:2a15a08b17dd94c53a1da0438822d70ebcd13f8c3a95abe3a9ef9f11a94830aa", size = 14571, upload-time = "2025-09-27T18:36:14.779Z" }, - { url = "https://files.pythonhosted.org/packages/d6/25/55dc3ab959917602c96985cb1253efaa4ff42f71194bddeb61eb7278b8be/markupsafe-3.0.3-cp310-cp310-win_amd64.whl", hash = "sha256:c4ffb7ebf07cfe8931028e3e4c85f0357459a3f9f9490886198848f4fa002ec8", size = 15056, upload-time = "2025-09-27T18:36:16.125Z" }, - { url = "https://files.pythonhosted.org/packages/d0/9e/0a02226640c255d1da0b8d12e24ac2aa6734da68bff14c05dd53b94a0fc3/markupsafe-3.0.3-cp310-cp310-win_arm64.whl", hash = "sha256:e2103a929dfa2fcaf9bb4e7c091983a49c9ac3b19c9061b6d5427dd7d14d81a1", size = 13932, upload-time = "2025-09-27T18:36:17.311Z" }, { url = "https://files.pythonhosted.org/packages/08/db/fefacb2136439fc8dd20e797950e749aa1f4997ed584c62cfb8ef7c2be0e/markupsafe-3.0.3-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:1cc7ea17a6824959616c525620e387f6dd30fec8cb44f649e31712db02123dad", size = 11631, upload-time = "2025-09-27T18:36:18.185Z" }, { url = "https://files.pythonhosted.org/packages/e1/2e/5898933336b61975ce9dc04decbc0a7f2fee78c30353c5efba7f2d6ff27a/markupsafe-3.0.3-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:4bd4cd07944443f5a265608cc6aab442e4f74dff8088b0dfc8238647b8f6ae9a", size = 12058, upload-time = "2025-09-27T18:36:19.444Z" }, { url = "https://files.pythonhosted.org/packages/1d/09/adf2df3699d87d1d8184038df46a9c80d78c0148492323f4693df54e17bb/markupsafe-3.0.3-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:6b5420a1d9450023228968e7e6a9ce57f65d148ab56d2313fcd589eee96a7a50", size = 24287, upload-time = "2025-09-27T18:36:20.768Z" }, @@ -877,17 +804,10 @@ dependencies = [ { name = "librt", marker = "platform_python_implementation != 'PyPy'" }, { name = "mypy-extensions" }, { name = "pathspec" }, - { name = "tomli", marker = "python_full_version < '3.11'" }, { name = "typing-extensions" }, ] sdist = { url = "https://files.pythonhosted.org/packages/f5/db/4efed9504bc01309ab9c2da7e352cc223569f05478012b5d9ece38fd44d2/mypy-1.19.1.tar.gz", hash = "sha256:19d88bb05303fe63f71dd2c6270daca27cb9401c4ca8255fe50d1d920e0eb9ba", size = 3582404, upload-time = "2025-12-15T05:03:48.42Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/2f/63/e499890d8e39b1ff2df4c0c6ce5d371b6844ee22b8250687a99fd2f657a8/mypy-1.19.1-cp310-cp310-macosx_10_9_x86_64.whl", hash = "sha256:5f05aa3d375b385734388e844bc01733bd33c644ab48e9684faa54e5389775ec", size = 13101333, upload-time = "2025-12-15T05:03:03.28Z" }, - { url = "https://files.pythonhosted.org/packages/72/4b/095626fc136fba96effc4fd4a82b41d688ab92124f8c4f7564bffe5cf1b0/mypy-1.19.1-cp310-cp310-macosx_11_0_arm64.whl", hash = "sha256:022ea7279374af1a5d78dfcab853fe6a536eebfda4b59deab53cd21f6cd9f00b", size = 12164102, upload-time = "2025-12-15T05:02:33.611Z" }, - { url = "https://files.pythonhosted.org/packages/0c/5b/952928dd081bf88a83a5ccd49aaecfcd18fd0d2710c7ff07b8fb6f7032b9/mypy-1.19.1-cp310-cp310-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:ee4c11e460685c3e0c64a4c5de82ae143622410950d6be863303a1c4ba0e36d6", size = 12765799, upload-time = "2025-12-15T05:03:28.44Z" }, - { url = "https://files.pythonhosted.org/packages/2a/0d/93c2e4a287f74ef11a66fb6d49c7a9f05e47b0a4399040e6719b57f500d2/mypy-1.19.1-cp310-cp310-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:de759aafbae8763283b2ee5869c7255391fbc4de3ff171f8f030b5ec48381b74", size = 13522149, upload-time = "2025-12-15T05:02:36.011Z" }, - { url = "https://files.pythonhosted.org/packages/7b/0e/33a294b56aaad2b338d203e3a1d8b453637ac36cb278b45005e0901cf148/mypy-1.19.1-cp310-cp310-musllinux_1_2_x86_64.whl", hash = "sha256:ab43590f9cd5108f41aacf9fca31841142c786827a74ab7cc8a2eacb634e09a1", size = 13810105, upload-time = "2025-12-15T05:02:40.327Z" }, - { url = "https://files.pythonhosted.org/packages/0e/fd/3e82603a0cb66b67c5e7abababce6bf1a929ddf67bf445e652684af5c5a0/mypy-1.19.1-cp310-cp310-win_amd64.whl", hash = "sha256:2899753e2f61e571b3971747e302d5f420c3fd09650e1951e99f823bc3089dac", size = 10057200, upload-time = "2025-12-15T05:02:51.012Z" }, { url = "https://files.pythonhosted.org/packages/ef/47/6b3ebabd5474d9cdc170d1342fbf9dddc1b0ec13ec90bf9004ee6f391c31/mypy-1.19.1-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:d8dfc6ab58ca7dda47d9237349157500468e404b17213d44fc1cb77bce532288", size = 13028539, upload-time = "2025-12-15T05:03:44.129Z" }, { url = "https://files.pythonhosted.org/packages/5c/a6/ac7c7a88a3c9c54334f53a941b765e6ec6c4ebd65d3fe8cdcfbe0d0fd7db/mypy-1.19.1-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:e3f276d8493c3c97930e354b2595a44a21348b320d859fb4a2b9f66da9ed27ab", size = 12083163, upload-time = "2025-12-15T05:03:37.679Z" }, { url = "https://files.pythonhosted.org/packages/67/af/3afa9cf880aa4a2c803798ac24f1d11ef72a0c8079689fac5cfd815e2830/mypy-1.19.1-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:2abb24cf3f17864770d18d673c85235ba52456b36a06b6afc1e07c1fdcd3d0e6", size = 12687629, upload-time = "2025-12-15T05:02:31.526Z" }, @@ -1098,19 +1018,6 @@ version = "3.11.7" source = { registry = "https://pypi.org/simple" } sdist = { url = "https://files.pythonhosted.org/packages/53/45/b268004f745ede84e5798b48ee12b05129d19235d0e15267aa57dcdb400b/orjson-3.11.7.tar.gz", hash = "sha256:9b1a67243945819ce55d24a30b59d6a168e86220452d2c96f4d1f093e71c0c49", size = 6144992, upload-time = "2026-02-02T15:38:49.29Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/de/1a/a373746fa6d0e116dd9e54371a7b54622c44d12296d5d0f3ad5e3ff33490/orjson-3.11.7-cp310-cp310-macosx_10_15_x86_64.macosx_11_0_arm64.macosx_10_15_universal2.whl", hash = "sha256:a02c833f38f36546ba65a452127633afce4cf0dd7296b753d3bb54e55e5c0174", size = 229140, upload-time = "2026-02-02T15:37:06.082Z" }, - { url = "https://files.pythonhosted.org/packages/52/a2/fa129e749d500f9b183e8a3446a193818a25f60261e9ce143ad61e975208/orjson-3.11.7-cp310-cp310-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:b63c6e6738d7c3470ad01601e23376aa511e50e1f3931395b9f9c722406d1a67", size = 128670, upload-time = "2026-02-02T15:37:08.002Z" }, - { url = "https://files.pythonhosted.org/packages/08/93/1e82011cd1e0bd051ef9d35bed1aa7fb4ea1f0a055dc2c841b46b43a9ebd/orjson-3.11.7-cp310-cp310-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:043d3006b7d32c7e233b8cfb1f01c651013ea079e08dcef7189a29abd8befe11", size = 123832, upload-time = "2026-02-02T15:37:09.191Z" }, - { url = "https://files.pythonhosted.org/packages/fe/d8/a26b431ef962c7d55736674dddade876822f3e33223c1f47a36879350d04/orjson-3.11.7-cp310-cp310-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:57036b27ac8a25d81112eb0cc9835cd4833c5b16e1467816adc0015f59e870dc", size = 129171, upload-time = "2026-02-02T15:37:11.112Z" }, - { url = "https://files.pythonhosted.org/packages/a7/19/f47819b84a580f490da260c3ee9ade214cf4cf78ac9ce8c1c758f80fdfc9/orjson-3.11.7-cp310-cp310-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:733ae23ada68b804b222c44affed76b39e30806d38660bf1eb200520d259cc16", size = 141967, upload-time = "2026-02-02T15:37:12.282Z" }, - { url = "https://files.pythonhosted.org/packages/5b/cd/37ece39a0777ba077fdcdbe4cccae3be8ed00290c14bf8afdc548befc260/orjson-3.11.7-cp310-cp310-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:5fdfad2093bdd08245f2e204d977facd5f871c88c4a71230d5bcbd0e43bf6222", size = 130991, upload-time = "2026-02-02T15:37:13.465Z" }, - { url = "https://files.pythonhosted.org/packages/8f/ed/f2b5d66aa9b6b5c02ff5f120efc7b38c7c4962b21e6be0f00fd99a5c348e/orjson-3.11.7-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:cededd6738e1c153530793998e31c05086582b08315db48ab66649768f326baa", size = 133674, upload-time = "2026-02-02T15:37:14.694Z" }, - { url = "https://files.pythonhosted.org/packages/c4/6e/baa83e68d1aa09fa8c3e5b2c087d01d0a0bd45256de719ed7bc22c07052d/orjson-3.11.7-cp310-cp310-musllinux_1_2_aarch64.whl", hash = "sha256:14f440c7268c8f8633d1b3d443a434bd70cb15686117ea6beff8fdc8f5917a1e", size = 138722, upload-time = "2026-02-02T15:37:16.501Z" }, - { url = "https://files.pythonhosted.org/packages/0c/47/7f8ef4963b772cd56999b535e553f7eb5cd27e9dd6c049baee6f18bfa05d/orjson-3.11.7-cp310-cp310-musllinux_1_2_armv7l.whl", hash = "sha256:3a2479753bbb95b0ebcf7969f562cdb9668e6d12416a35b0dda79febf89cdea2", size = 409056, upload-time = "2026-02-02T15:37:17.895Z" }, - { url = "https://files.pythonhosted.org/packages/38/eb/2df104dd2244b3618f25325a656f85cc3277f74bbd91224752410a78f3c7/orjson-3.11.7-cp310-cp310-musllinux_1_2_i686.whl", hash = "sha256:71924496986275a737f38e3f22b4e0878882b3f7a310d2ff4dc96e812789120c", size = 144196, upload-time = "2026-02-02T15:37:19.349Z" }, - { url = "https://files.pythonhosted.org/packages/b6/2a/ee41de0aa3a6686598661eae2b4ebdff1340c65bfb17fcff8b87138aab21/orjson-3.11.7-cp310-cp310-musllinux_1_2_x86_64.whl", hash = "sha256:b4a9eefdc70bf8bf9857f0290f973dec534ac84c35cd6a7f4083be43e7170a8f", size = 134979, upload-time = "2026-02-02T15:37:20.906Z" }, - { url = "https://files.pythonhosted.org/packages/4c/fa/92fc5d3d402b87a8b28277a9ed35386218a6a5287c7fe5ee9b9f02c53fb2/orjson-3.11.7-cp310-cp310-win32.whl", hash = "sha256:ae9e0b37a834cef7ce8f99de6498f8fad4a2c0bf6bfc3d02abd8ed56aa15b2de", size = 127968, upload-time = "2026-02-02T15:37:23.178Z" }, - { url = "https://files.pythonhosted.org/packages/07/29/a576bf36d73d60df06904d3844a9df08e25d59eba64363aaf8ec2f9bff41/orjson-3.11.7-cp310-cp310-win_amd64.whl", hash = "sha256:d772afdb22555f0c58cfc741bdae44180122b3616faa1ecadb595cd526e4c993", size = 125128, upload-time = "2026-02-02T15:37:24.329Z" }, { url = "https://files.pythonhosted.org/packages/37/02/da6cb01fc6087048d7f61522c327edf4250f1683a58a839fdcc435746dd5/orjson-3.11.7-cp311-cp311-macosx_10_15_x86_64.macosx_11_0_arm64.macosx_10_15_universal2.whl", hash = "sha256:9487abc2c2086e7c8eb9a211d2ce8855bae0e92586279d0d27b341d5ad76c85c", size = 228664, upload-time = "2026-02-02T15:37:25.542Z" }, { url = "https://files.pythonhosted.org/packages/c1/c2/5885e7a5881dba9a9af51bc564e8967225a642b3e03d089289a35054e749/orjson-3.11.7-cp311-cp311-macosx_15_0_arm64.whl", hash = "sha256:79cacb0b52f6004caf92405a7e1f11e6e2de8bdf9019e4f76b44ba045125cd6b", size = 125344, upload-time = "2026-02-02T15:37:26.92Z" }, { url = "https://files.pythonhosted.org/packages/a4/1d/4e7688de0a92d1caf600dfd5fb70b4c5bfff51dfa61ac555072ef2d0d32a/orjson-3.11.7-cp311-cp311-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:c2e85fe4698b6a56d5e2ebf7ae87544d668eb6bde1ad1226c13f44663f20ec9e", size = 128404, upload-time = "2026-02-02T15:37:28.108Z" }, @@ -1179,14 +1086,6 @@ version = "1.12.2" source = { registry = "https://pypi.org/simple" } sdist = { url = "https://files.pythonhosted.org/packages/12/0c/f1761e21486942ab9bb6feaebc610fa074f7c5e496e6962dea5873348077/ormsgpack-1.12.2.tar.gz", hash = "sha256:944a2233640273bee67521795a73cf1e959538e0dfb7ac635505010455e53b33", size = 39031, upload-time = "2026-01-18T20:55:28.023Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/93/fa/a91f70829ebccf6387c4946e0a1a109f6ba0d6a28d65f628bedfad94b890/ormsgpack-1.12.2-cp310-cp310-macosx_10_12_x86_64.macosx_11_0_arm64.macosx_10_12_universal2.whl", hash = "sha256:c1429217f8f4d7fcb053523bbbac6bed5e981af0b85ba616e6df7cce53c19657", size = 378262, upload-time = "2026-01-18T20:55:22.284Z" }, - { url = "https://files.pythonhosted.org/packages/5f/62/3698a9a0c487252b5c6a91926e5654e79e665708ea61f67a8bdeceb022bf/ormsgpack-1.12.2-cp310-cp310-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:5f13034dc6c84a6280c6c33db7ac420253852ea233fc3ee27c8875f8dd651163", size = 203034, upload-time = "2026-01-18T20:55:53.324Z" }, - { url = "https://files.pythonhosted.org/packages/66/3a/f716f64edc4aec2744e817660b317e2f9bb8de372338a95a96198efa1ac1/ormsgpack-1.12.2-cp310-cp310-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:59f5da97000c12bc2d50e988bdc8576b21f6ab4e608489879d35b2c07a8ab51a", size = 210538, upload-time = "2026-01-18T20:55:20.097Z" }, - { url = "https://files.pythonhosted.org/packages/72/30/a436be9ce27d693d4e19fa94900028067133779f09fc45776db3f689c822/ormsgpack-1.12.2-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:9e4459c3f27066beadb2b81ea48a076a417aafffff7df1d3c11c519190ed44f2", size = 212401, upload-time = "2026-01-18T20:55:46.447Z" }, - { url = "https://files.pythonhosted.org/packages/10/c5/cde98300fd33fee84ca71de4751b19aeeca675f0cf3c0ec4b043f40f3b76/ormsgpack-1.12.2-cp310-cp310-musllinux_1_2_aarch64.whl", hash = "sha256:7a1c460655d7288407ffa09065e322a7231997c0d62ce914bf3a96ad2dc6dedd", size = 387080, upload-time = "2026-01-18T20:56:00.884Z" }, - { url = "https://files.pythonhosted.org/packages/6a/31/30bf445ef827546747c10889dd254b3d84f92b591300efe4979d792f4c41/ormsgpack-1.12.2-cp310-cp310-musllinux_1_2_armv7l.whl", hash = "sha256:458e4568be13d311ef7d8877275e7ccbe06c0e01b39baaac874caaa0f46d826c", size = 482346, upload-time = "2026-01-18T20:55:39.831Z" }, - { url = "https://files.pythonhosted.org/packages/2e/f5/e1745ddf4fa246c921b5ca253636c4c700ff768d78032f79171289159f6e/ormsgpack-1.12.2-cp310-cp310-musllinux_1_2_x86_64.whl", hash = "sha256:8cde5eaa6c6cbc8622db71e4a23de56828e3d876aeb6460ffbcb5b8aff91093b", size = 425178, upload-time = "2026-01-18T20:55:27.106Z" }, - { url = "https://files.pythonhosted.org/packages/8d/a2/e6532ed7716aed03dede8df2d0d0d4150710c2122647d94b474147ccd891/ormsgpack-1.12.2-cp310-cp310-win_amd64.whl", hash = "sha256:dc7a33be14c347893edbb1ceda89afbf14c467d593a5ee92c11de4f1666b4d4f", size = 117183, upload-time = "2026-01-18T20:55:55.52Z" }, { url = "https://files.pythonhosted.org/packages/4b/08/8b68f24b18e69d92238aa8f258218e6dfeacf4381d9d07ab8df303f524a9/ormsgpack-1.12.2-cp311-cp311-macosx_10_12_x86_64.macosx_11_0_arm64.macosx_10_12_universal2.whl", hash = "sha256:bd5f4bf04c37888e864f08e740c5a573c4017f6fd6e99fa944c5c935fabf2dd9", size = 378266, upload-time = "2026-01-18T20:55:59.876Z" }, { url = "https://files.pythonhosted.org/packages/0d/24/29fc13044ecb7c153523ae0a1972269fcd613650d1fa1a9cec1044c6b666/ormsgpack-1.12.2-cp311-cp311-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:34d5b28b3570e9fed9a5a76528fc7230c3c76333bc214798958e58e9b79cc18a", size = 203035, upload-time = "2026-01-18T20:55:30.59Z" }, { url = "https://files.pythonhosted.org/packages/ad/c2/00169fb25dd8f9213f5e8a549dfb73e4d592009ebc85fbbcd3e1dcac575b/ormsgpack-1.12.2-cp311-cp311-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:3708693412c28f3538fb5a65da93787b6bbab3484f6bc6e935bfb77a62400ae5", size = 210539, upload-time = "2026-01-18T20:55:48.569Z" }, @@ -1285,15 +1184,6 @@ version = "0.11.0" source = { registry = "https://pypi.org/simple" } sdist = { url = "https://files.pythonhosted.org/packages/91/c7/e0b3bbe72e0003e5d02726e0d406ea47d523a2aec9c41d831817a8e0bce1/polyleven-0.11.0.tar.gz", hash = "sha256:d74d348387cf340051711c0dd6af993b4c264daa78470098de16f4a2b725785c", size = 6407, upload-time = "2026-02-09T09:41:49.87Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/dc/6c/34c6189c80adf7575fb2daee38c4b836c154e416ab3c14d17afa7f88b9c3/polyleven-0.11.0-cp310-cp310-macosx_10_9_x86_64.whl", hash = "sha256:ccf87f6ac8d76aa4c48a4828becc0c19fd1589b14b20affe23e5e012be4fa64f", size = 7421, upload-time = "2026-02-09T09:40:33.243Z" }, - { url = "https://files.pythonhosted.org/packages/e6/21/b20d3c9f9b6bded43a0388037044f2bcb1add20fa9a758d1144d79e09d10/polyleven-0.11.0-cp310-cp310-macosx_11_0_arm64.whl", hash = "sha256:c1a02e3f0acfd1164cbaea25192398bc943ee9b93b9883a1fba9b2613d3616b0", size = 7514, upload-time = "2026-02-09T09:40:34.514Z" }, - { url = "https://files.pythonhosted.org/packages/c6/d8/60290fd8d8298671edb4ce221d8ee4d81156b3c21b1154a615da7ea8e57d/polyleven-0.11.0-cp310-cp310-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:6526d2516b439065864722069de6fcc418a4135696990dad66b81ddb18863bd9", size = 19556, upload-time = "2026-02-09T09:40:35.868Z" }, - { url = "https://files.pythonhosted.org/packages/c1/cd/17f1f6009a344c18f4213edcc97a9e7dae22e9a26e722aae10052378a43e/polyleven-0.11.0-cp310-cp310-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:65fc01fe6cfe287f2f20170b35687a436ab36b882db568a55d81d6e0acd8379d", size = 20303, upload-time = "2026-02-09T09:40:36.99Z" }, - { url = "https://files.pythonhosted.org/packages/4e/18/30c7da8056adc4b9a81775f5b1810a2c9b6cc87fc34c8cf4c0280d28cab6/polyleven-0.11.0-cp310-cp310-musllinux_1_2_aarch64.whl", hash = "sha256:4b8e5ac8faecc6daa7b3d325436a3f23f8c33dec7bfca5d22df3fbe00f92ddd9", size = 19565, upload-time = "2026-02-09T09:40:38.471Z" }, - { url = "https://files.pythonhosted.org/packages/de/14/86e9c33ff9fda84297556373a6376100cbe9bb5d917fc3421dce4ada441b/polyleven-0.11.0-cp310-cp310-musllinux_1_2_x86_64.whl", hash = "sha256:dc4f17007b07fde292ad33ec43a3ae8febe27a5bd92462b920736fd81d774fce", size = 19365, upload-time = "2026-02-09T09:40:39.893Z" }, - { url = "https://files.pythonhosted.org/packages/be/18/1341f7860bbe2287f6cb8a540c3435c504cfc58659b67106ab60f695175e/polyleven-0.11.0-cp310-cp310-win32.whl", hash = "sha256:cae70197d545a09bfab8d7e506eed66ef314fa6c4e7a5e2c402c2febc31db74b", size = 11672, upload-time = "2026-02-09T09:40:40.822Z" }, - { url = "https://files.pythonhosted.org/packages/6d/09/ed2cb3dbec7a925d80107516b06dfe10dd368c6abfb43765df6feb6cd551/polyleven-0.11.0-cp310-cp310-win_amd64.whl", hash = "sha256:bf47079b6dc62e6af2bd6ecb45a6087efd9a27b61666b98d0326c246a22ea991", size = 10828, upload-time = "2026-02-09T09:40:42.224Z" }, - { url = "https://files.pythonhosted.org/packages/13/7a/cb74c2ffe4e35935d80ef6f180f9e0987b8000917d52050a3563b32e1e73/polyleven-0.11.0-cp310-cp310-win_arm64.whl", hash = "sha256:4c78b4d3e7d7b74315d5422178118963374c0cf3d7a9532a955f446ed365320a", size = 9389, upload-time = "2026-02-09T09:40:43.172Z" }, { url = "https://files.pythonhosted.org/packages/6c/7a/27ea9a78b617ddb14c2f5d2416df2fbf07fa5e52685f2968686a0308c8af/polyleven-0.11.0-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:a28860fe33a7f907bc5f86e55a0b9faea80047d1677fa23b4d6c631ccf91ef2f", size = 7420, upload-time = "2026-02-09T09:40:44.505Z" }, { url = "https://files.pythonhosted.org/packages/a6/9c/fea309d41502aa5a344a6d4d6e5b8bdabb1df1e28f1af52bb53180f6c956/polyleven-0.11.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:47a3fb5b8cb60f647d2832d38b7d87cda27da8622b27c1292bceb9a04954c189", size = 7514, upload-time = "2026-02-09T09:40:46.44Z" }, { url = "https://files.pythonhosted.org/packages/48/77/c7d3bb6c66050304c3fe3cae1a716f62fea947ac3f14d02ef71e24422f76/polyleven-0.11.0-cp311-cp311-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:209fa669ca23ac453a7e9fbf07695350d5cbe61d71a6226b861757ccab28e664", size = 20887, upload-time = "2026-02-09T09:40:47.304Z" }, @@ -1396,19 +1286,6 @@ dependencies = [ ] sdist = { url = "https://files.pythonhosted.org/packages/71/70/23b021c950c2addd24ec408e9ab05d59b035b39d97cdc1130e1bce647bb6/pydantic_core-2.41.5.tar.gz", hash = "sha256:08daa51ea16ad373ffd5e7606252cc32f07bc72b28284b6bc9c6df804816476e", size = 460952, upload-time = "2025-11-04T13:43:49.098Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/c6/90/32c9941e728d564b411d574d8ee0cf09b12ec978cb22b294995bae5549a5/pydantic_core-2.41.5-cp310-cp310-macosx_10_12_x86_64.whl", hash = "sha256:77b63866ca88d804225eaa4af3e664c5faf3568cea95360d21f4725ab6e07146", size = 2107298, upload-time = "2025-11-04T13:39:04.116Z" }, - { url = "https://files.pythonhosted.org/packages/fb/a8/61c96a77fe28993d9a6fb0f4127e05430a267b235a124545d79fea46dd65/pydantic_core-2.41.5-cp310-cp310-macosx_11_0_arm64.whl", hash = "sha256:dfa8a0c812ac681395907e71e1274819dec685fec28273a28905df579ef137e2", size = 1901475, upload-time = "2025-11-04T13:39:06.055Z" }, - { url = "https://files.pythonhosted.org/packages/5d/b6/338abf60225acc18cdc08b4faef592d0310923d19a87fba1faf05af5346e/pydantic_core-2.41.5-cp310-cp310-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:5921a4d3ca3aee735d9fd163808f5e8dd6c6972101e4adbda9a4667908849b97", size = 1918815, upload-time = "2025-11-04T13:39:10.41Z" }, - { url = "https://files.pythonhosted.org/packages/d1/1c/2ed0433e682983d8e8cba9c8d8ef274d4791ec6a6f24c58935b90e780e0a/pydantic_core-2.41.5-cp310-cp310-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:e25c479382d26a2a41b7ebea1043564a937db462816ea07afa8a44c0866d52f9", size = 2065567, upload-time = "2025-11-04T13:39:12.244Z" }, - { url = "https://files.pythonhosted.org/packages/b3/24/cf84974ee7d6eae06b9e63289b7b8f6549d416b5c199ca2d7ce13bbcf619/pydantic_core-2.41.5-cp310-cp310-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:f547144f2966e1e16ae626d8ce72b4cfa0caedc7fa28052001c94fb2fcaa1c52", size = 2230442, upload-time = "2025-11-04T13:39:13.962Z" }, - { url = "https://files.pythonhosted.org/packages/fd/21/4e287865504b3edc0136c89c9c09431be326168b1eb7841911cbc877a995/pydantic_core-2.41.5-cp310-cp310-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:6f52298fbd394f9ed112d56f3d11aabd0d5bd27beb3084cc3d8ad069483b8941", size = 2350956, upload-time = "2025-11-04T13:39:15.889Z" }, - { url = "https://files.pythonhosted.org/packages/a8/76/7727ef2ffa4b62fcab916686a68a0426b9b790139720e1934e8ba797e238/pydantic_core-2.41.5-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:100baa204bb412b74fe285fb0f3a385256dad1d1879f0a5cb1499ed2e83d132a", size = 2068253, upload-time = "2025-11-04T13:39:17.403Z" }, - { url = "https://files.pythonhosted.org/packages/d5/8c/a4abfc79604bcb4c748e18975c44f94f756f08fb04218d5cb87eb0d3a63e/pydantic_core-2.41.5-cp310-cp310-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:05a2c8852530ad2812cb7914dc61a1125dc4e06252ee98e5638a12da6cc6fb6c", size = 2177050, upload-time = "2025-11-04T13:39:19.351Z" }, - { url = "https://files.pythonhosted.org/packages/67/b1/de2e9a9a79b480f9cb0b6e8b6ba4c50b18d4e89852426364c66aa82bb7b3/pydantic_core-2.41.5-cp310-cp310-musllinux_1_1_aarch64.whl", hash = "sha256:29452c56df2ed968d18d7e21f4ab0ac55e71dc59524872f6fc57dcf4a3249ed2", size = 2147178, upload-time = "2025-11-04T13:39:21Z" }, - { url = "https://files.pythonhosted.org/packages/16/c1/dfb33f837a47b20417500efaa0378adc6635b3c79e8369ff7a03c494b4ac/pydantic_core-2.41.5-cp310-cp310-musllinux_1_1_armv7l.whl", hash = "sha256:d5160812ea7a8a2ffbe233d8da666880cad0cbaf5d4de74ae15c313213d62556", size = 2341833, upload-time = "2025-11-04T13:39:22.606Z" }, - { url = "https://files.pythonhosted.org/packages/47/36/00f398642a0f4b815a9a558c4f1dca1b4020a7d49562807d7bc9ff279a6c/pydantic_core-2.41.5-cp310-cp310-musllinux_1_1_x86_64.whl", hash = "sha256:df3959765b553b9440adfd3c795617c352154e497a4eaf3752555cfb5da8fc49", size = 2321156, upload-time = "2025-11-04T13:39:25.843Z" }, - { url = "https://files.pythonhosted.org/packages/7e/70/cad3acd89fde2010807354d978725ae111ddf6d0ea46d1ea1775b5c1bd0c/pydantic_core-2.41.5-cp310-cp310-win32.whl", hash = "sha256:1f8d33a7f4d5a7889e60dc39856d76d09333d8a6ed0f5f1190635cbec70ec4ba", size = 1989378, upload-time = "2025-11-04T13:39:27.92Z" }, - { url = "https://files.pythonhosted.org/packages/76/92/d338652464c6c367e5608e4488201702cd1cbb0f33f7b6a85a60fe5f3720/pydantic_core-2.41.5-cp310-cp310-win_amd64.whl", hash = "sha256:62de39db01b8d593e45871af2af9e497295db8d73b085f6bfd0b18c83c70a8f9", size = 2013622, upload-time = "2025-11-04T13:39:29.848Z" }, { url = "https://files.pythonhosted.org/packages/e8/72/74a989dd9f2084b3d9530b0915fdda64ac48831c30dbf7c72a41a5232db8/pydantic_core-2.41.5-cp311-cp311-macosx_10_12_x86_64.whl", hash = "sha256:a3a52f6156e73e7ccb0f8cced536adccb7042be67cb45f9562e12b319c119da6", size = 2105873, upload-time = "2025-11-04T13:39:31.373Z" }, { url = "https://files.pythonhosted.org/packages/12/44/37e403fd9455708b3b942949e1d7febc02167662bf1a7da5b78ee1ea2842/pydantic_core-2.41.5-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:7f3bf998340c6d4b0c9a2f02d6a400e51f123b59565d74dc60d252ce888c260b", size = 1899826, upload-time = "2025-11-04T13:39:32.897Z" }, { url = "https://files.pythonhosted.org/packages/33/7f/1d5cab3ccf44c1935a359d51a8a2a9e1a654b744b5e7f80d41b88d501eec/pydantic_core-2.41.5-cp311-cp311-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:378bec5c66998815d224c9ca994f1e14c0c21cb95d2f52b6021cc0b2a58f2a5a", size = 1917869, upload-time = "2025-11-04T13:39:34.469Z" }, @@ -1487,14 +1364,6 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/aa/81/05e400037eaf55ad400bcd318c05bb345b57e708887f07ddb2d20e3f0e98/pydantic_core-2.41.5-graalpy312-graalpy250_312_native-macosx_11_0_arm64.whl", hash = "sha256:aabf5777b5c8ca26f7824cb4a120a740c9588ed58df9b2d196ce92fba42ff8dc", size = 1915388, upload-time = "2025-11-04T13:42:52.215Z" }, { url = "https://files.pythonhosted.org/packages/6e/0d/e3549b2399f71d56476b77dbf3cf8937cec5cd70536bdc0e374a421d0599/pydantic_core-2.41.5-graalpy312-graalpy250_312_native-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:c007fe8a43d43b3969e8469004e9845944f1a80e6acd47c150856bb87f230c56", size = 1942879, upload-time = "2025-11-04T13:42:56.483Z" }, { url = "https://files.pythonhosted.org/packages/f7/07/34573da085946b6a313d7c42f82f16e8920bfd730665de2d11c0c37a74b5/pydantic_core-2.41.5-graalpy312-graalpy250_312_native-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:76d0819de158cd855d1cbb8fcafdf6f5cf1eb8e470abe056d5d161106e38062b", size = 2139017, upload-time = "2025-11-04T13:42:59.471Z" }, - { url = "https://files.pythonhosted.org/packages/e6/b0/1a2aa41e3b5a4ba11420aba2d091b2d17959c8d1519ece3627c371951e73/pydantic_core-2.41.5-pp310-pypy310_pp73-macosx_10_12_x86_64.whl", hash = "sha256:b5819cd790dbf0c5eb9f82c73c16b39a65dd6dd4d1439dcdea7816ec9adddab8", size = 2103351, upload-time = "2025-11-04T13:43:02.058Z" }, - { url = "https://files.pythonhosted.org/packages/a4/ee/31b1f0020baaf6d091c87900ae05c6aeae101fa4e188e1613c80e4f1ea31/pydantic_core-2.41.5-pp310-pypy310_pp73-macosx_11_0_arm64.whl", hash = "sha256:5a4e67afbc95fa5c34cf27d9089bca7fcab4e51e57278d710320a70b956d1b9a", size = 1925363, upload-time = "2025-11-04T13:43:05.159Z" }, - { url = "https://files.pythonhosted.org/packages/e1/89/ab8e86208467e467a80deaca4e434adac37b10a9d134cd2f99b28a01e483/pydantic_core-2.41.5-pp310-pypy310_pp73-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:ece5c59f0ce7d001e017643d8d24da587ea1f74f6993467d85ae8a5ef9d4f42b", size = 2135615, upload-time = "2025-11-04T13:43:08.116Z" }, - { url = "https://files.pythonhosted.org/packages/99/0a/99a53d06dd0348b2008f2f30884b34719c323f16c3be4e6cc1203b74a91d/pydantic_core-2.41.5-pp310-pypy310_pp73-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:16f80f7abe3351f8ea6858914ddc8c77e02578544a0ebc15b4c2e1a0e813b0b2", size = 2175369, upload-time = "2025-11-04T13:43:12.49Z" }, - { url = "https://files.pythonhosted.org/packages/6d/94/30ca3b73c6d485b9bb0bc66e611cff4a7138ff9736b7e66bcf0852151636/pydantic_core-2.41.5-pp310-pypy310_pp73-musllinux_1_1_aarch64.whl", hash = "sha256:33cb885e759a705b426baada1fe68cbb0a2e68e34c5d0d0289a364cf01709093", size = 2144218, upload-time = "2025-11-04T13:43:15.431Z" }, - { url = "https://files.pythonhosted.org/packages/87/57/31b4f8e12680b739a91f472b5671294236b82586889ef764b5fbc6669238/pydantic_core-2.41.5-pp310-pypy310_pp73-musllinux_1_1_armv7l.whl", hash = "sha256:c8d8b4eb992936023be7dee581270af5c6e0697a8559895f527f5b7105ecd36a", size = 2329951, upload-time = "2025-11-04T13:43:18.062Z" }, - { url = "https://files.pythonhosted.org/packages/7d/73/3c2c8edef77b8f7310e6fb012dbc4b8551386ed575b9eb6fb2506e28a7eb/pydantic_core-2.41.5-pp310-pypy310_pp73-musllinux_1_1_x86_64.whl", hash = "sha256:242a206cd0318f95cd21bdacff3fcc3aab23e79bba5cac3db5a841c9ef9c6963", size = 2318428, upload-time = "2025-11-04T13:43:20.679Z" }, - { url = "https://files.pythonhosted.org/packages/2f/02/8559b1f26ee0d502c74f9cca5c0d2fd97e967e083e006bbbb4e97f3a043a/pydantic_core-2.41.5-pp310-pypy310_pp73-win_amd64.whl", hash = "sha256:d3a978c4f57a597908b7e697229d996d77a6d3c94901e9edee593adada95ce1a", size = 2147009, upload-time = "2025-11-04T13:43:23.286Z" }, { url = "https://files.pythonhosted.org/packages/5f/9b/1b3f0e9f9305839d7e84912f9e8bfbd191ed1b1ef48083609f0dabde978c/pydantic_core-2.41.5-pp311-pypy311_pp73-macosx_10_12_x86_64.whl", hash = "sha256:b2379fa7ed44ddecb5bfe4e48577d752db9fc10be00a6b7446e9663ba143de26", size = 2101980, upload-time = "2025-11-04T13:43:25.97Z" }, { url = "https://files.pythonhosted.org/packages/a4/ed/d71fefcb4263df0da6a85b5d8a7508360f2f2e9b3bf5814be9c8bccdccc1/pydantic_core-2.41.5-pp311-pypy311_pp73-macosx_11_0_arm64.whl", hash = "sha256:266fb4cbf5e3cbd0b53669a6d1b039c45e3ce651fd5442eff4d07c2cc8d66808", size = 1923865, upload-time = "2025-11-04T13:43:28.763Z" }, { url = "https://files.pythonhosted.org/packages/ce/3a/626b38db460d675f873e4444b4bb030453bbe7b4ba55df821d026a0493c4/pydantic_core-2.41.5-pp311-pypy311_pp73-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:58133647260ea01e4d0500089a8c4f07bd7aa6ce109682b1426394988d8aaacc", size = 2134256, upload-time = "2025-11-04T13:43:31.71Z" }, @@ -1520,12 +1389,10 @@ version = "8.4.2" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "colorama", marker = "sys_platform == 'win32'" }, - { name = "exceptiongroup", marker = "python_full_version < '3.11'" }, { name = "iniconfig" }, { name = "packaging" }, { name = "pluggy" }, { name = "pygments" }, - { name = "tomli", marker = "python_full_version < '3.11'" }, ] sdist = { url = "https://files.pythonhosted.org/packages/a3/5c/00a0e072241553e1a7496d638deababa67c5058571567b92a7eaa258397c/pytest-8.4.2.tar.gz", hash = "sha256:86c0d0b93306b961d58d62a4db4879f27fe25513d4b969df351abdddb3c30e01", size = 1519618, upload-time = "2025-09-04T14:34:22.711Z" } wheels = [ @@ -1537,7 +1404,6 @@ name = "pytest-asyncio" version = "1.1.1" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "backports-asyncio-runner", marker = "python_full_version < '3.11'" }, { name = "pytest" }, ] sdist = { url = "https://files.pythonhosted.org/packages/8d/1e/2aa43805d4a320a9489d2b99f7877b69f9094c79aa0732159a1415dd6cd4/pytest_asyncio-1.1.1.tar.gz", hash = "sha256:b72d215c38e2c91dbb32f275e0b5be69602d7869910e109360e375129960a649", size = 46590, upload-time = "2025-09-12T06:36:20.834Z" } @@ -1601,15 +1467,6 @@ version = "6.0.3" source = { registry = "https://pypi.org/simple" } sdist = { url = "https://files.pythonhosted.org/packages/05/8e/961c0007c59b8dd7729d542c61a4d537767a59645b82a0b521206e1e25c2/pyyaml-6.0.3.tar.gz", hash = "sha256:d76623373421df22fb4cf8817020cbb7ef15c725b9d5e45f17e189bfc384190f", size = 130960, upload-time = "2025-09-25T21:33:16.546Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/f4/a0/39350dd17dd6d6c6507025c0e53aef67a9293a6d37d3511f23ea510d5800/pyyaml-6.0.3-cp310-cp310-macosx_10_13_x86_64.whl", hash = "sha256:214ed4befebe12df36bcc8bc2b64b396ca31be9304b8f59e25c11cf94a4c033b", size = 184227, upload-time = "2025-09-25T21:31:46.04Z" }, - { url = "https://files.pythonhosted.org/packages/05/14/52d505b5c59ce73244f59c7a50ecf47093ce4765f116cdb98286a71eeca2/pyyaml-6.0.3-cp310-cp310-macosx_11_0_arm64.whl", hash = "sha256:02ea2dfa234451bbb8772601d7b8e426c2bfa197136796224e50e35a78777956", size = 174019, upload-time = "2025-09-25T21:31:47.706Z" }, - { url = "https://files.pythonhosted.org/packages/43/f7/0e6a5ae5599c838c696adb4e6330a59f463265bfa1e116cfd1fbb0abaaae/pyyaml-6.0.3-cp310-cp310-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:b30236e45cf30d2b8e7b3e85881719e98507abed1011bf463a8fa23e9c3e98a8", size = 740646, upload-time = "2025-09-25T21:31:49.21Z" }, - { url = "https://files.pythonhosted.org/packages/2f/3a/61b9db1d28f00f8fd0ae760459a5c4bf1b941baf714e207b6eb0657d2578/pyyaml-6.0.3-cp310-cp310-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:66291b10affd76d76f54fad28e22e51719ef9ba22b29e1d7d03d6777a9174198", size = 840793, upload-time = "2025-09-25T21:31:50.735Z" }, - { url = "https://files.pythonhosted.org/packages/7a/1e/7acc4f0e74c4b3d9531e24739e0ab832a5edf40e64fbae1a9c01941cabd7/pyyaml-6.0.3-cp310-cp310-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:9c7708761fccb9397fe64bbc0395abcae8c4bf7b0eac081e12b809bf47700d0b", size = 770293, upload-time = "2025-09-25T21:31:51.828Z" }, - { url = "https://files.pythonhosted.org/packages/8b/ef/abd085f06853af0cd59fa5f913d61a8eab65d7639ff2a658d18a25d6a89d/pyyaml-6.0.3-cp310-cp310-musllinux_1_2_aarch64.whl", hash = "sha256:418cf3f2111bc80e0933b2cd8cd04f286338bb88bdc7bc8e6dd775ebde60b5e0", size = 732872, upload-time = "2025-09-25T21:31:53.282Z" }, - { url = "https://files.pythonhosted.org/packages/1f/15/2bc9c8faf6450a8b3c9fc5448ed869c599c0a74ba2669772b1f3a0040180/pyyaml-6.0.3-cp310-cp310-musllinux_1_2_x86_64.whl", hash = "sha256:5e0b74767e5f8c593e8c9b5912019159ed0533c70051e9cce3e8b6aa699fcd69", size = 758828, upload-time = "2025-09-25T21:31:54.807Z" }, - { url = "https://files.pythonhosted.org/packages/a3/00/531e92e88c00f4333ce359e50c19b8d1de9fe8d581b1534e35ccfbc5f393/pyyaml-6.0.3-cp310-cp310-win32.whl", hash = "sha256:28c8d926f98f432f88adc23edf2e6d4921ac26fb084b028c733d01868d19007e", size = 142415, upload-time = "2025-09-25T21:31:55.885Z" }, - { url = "https://files.pythonhosted.org/packages/2a/fa/926c003379b19fca39dd4634818b00dec6c62d87faf628d1394e137354d4/pyyaml-6.0.3-cp310-cp310-win_amd64.whl", hash = "sha256:bdb2c67c6c1390b63c6ff89f210c8fd09d9a1217a465701eac7316313c915e4c", size = 158561, upload-time = "2025-09-25T21:31:57.406Z" }, { url = "https://files.pythonhosted.org/packages/6d/16/a95b6757765b7b031c9374925bb718d55e0a9ba8a1b6a12d25962ea44347/pyyaml-6.0.3-cp311-cp311-macosx_10_13_x86_64.whl", hash = "sha256:44edc647873928551a01e7a563d7452ccdebee747728c1080d881d68af7b997e", size = 185826, upload-time = "2025-09-25T21:31:58.655Z" }, { url = "https://files.pythonhosted.org/packages/16/19/13de8e4377ed53079ee996e1ab0a9c33ec2faf808a4647b7b4c0d46dd239/pyyaml-6.0.3-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:652cb6edd41e718550aad172851962662ff2681490a8a711af6a4d288dd96824", size = 175577, upload-time = "2025-09-25T21:32:00.088Z" }, { url = "https://files.pythonhosted.org/packages/0c/62/d2eb46264d4b157dae1275b573017abec435397aa59cbcdab6fc978a8af4/pyyaml-6.0.3-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:10892704fc220243f5305762e276552a0395f7beb4dbf9b14ec8fd43b57f126c", size = 775556, upload-time = "2025-09-25T21:32:01.31Z" }, @@ -1679,23 +1536,6 @@ version = "2026.2.28" source = { registry = "https://pypi.org/simple" } sdist = { url = "https://files.pythonhosted.org/packages/8b/71/41455aa99a5a5ac1eaf311f5d8efd9ce6433c03ac1e0962de163350d0d97/regex-2026.2.28.tar.gz", hash = "sha256:a729e47d418ea11d03469f321aaf67cdee8954cde3ff2cf8403ab87951ad10f2", size = 415184, upload-time = "2026-02-28T02:19:42.792Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/70/b8/845a927e078f5e5cc55d29f57becbfde0003d52806544531ab3f2da4503c/regex-2026.2.28-cp310-cp310-macosx_10_9_universal2.whl", hash = "sha256:fc48c500838be6882b32748f60a15229d2dea96e59ef341eaa96ec83538f498d", size = 488461, upload-time = "2026-02-28T02:15:48.405Z" }, - { url = "https://files.pythonhosted.org/packages/32/f9/8a0034716684e38a729210ded6222249f29978b24b684f448162ef21f204/regex-2026.2.28-cp310-cp310-macosx_10_9_x86_64.whl", hash = "sha256:2afa673660928d0b63d84353c6c08a8a476ddfc4a47e11742949d182e6863ce8", size = 290774, upload-time = "2026-02-28T02:15:51.738Z" }, - { url = "https://files.pythonhosted.org/packages/a6/ba/b27feefffbb199528dd32667cd172ed484d9c197618c575f01217fbe6103/regex-2026.2.28-cp310-cp310-macosx_11_0_arm64.whl", hash = "sha256:7ab218076eb0944549e7fe74cf0e2b83a82edb27e81cc87411f76240865e04d5", size = 288737, upload-time = "2026-02-28T02:15:53.534Z" }, - { url = "https://files.pythonhosted.org/packages/18/c5/65379448ca3cbfe774fcc33774dc8295b1ee97dc3237ae3d3c7b27423c9d/regex-2026.2.28-cp310-cp310-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:94d63db12e45a9b9f064bfe4800cefefc7e5f182052e4c1b774d46a40ab1d9bb", size = 782675, upload-time = "2026-02-28T02:15:55.488Z" }, - { url = "https://files.pythonhosted.org/packages/aa/30/6fa55bef48090f900fbd4649333791fc3e6467380b9e775e741beeb3231f/regex-2026.2.28-cp310-cp310-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:195237dc327858a7721bf8b0bbbef797554bc13563c3591e91cd0767bacbe359", size = 850514, upload-time = "2026-02-28T02:15:57.509Z" }, - { url = "https://files.pythonhosted.org/packages/a9/28/9ca180fb3787a54150209754ac06a42409913571fa94994f340b3bba4e1e/regex-2026.2.28-cp310-cp310-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:b387a0d092dac157fb026d737dde35ff3e49ef27f285343e7c6401851239df27", size = 896612, upload-time = "2026-02-28T02:15:59.682Z" }, - { url = "https://files.pythonhosted.org/packages/46/b5/f30d7d3936d6deecc3ea7bea4f7d3c5ee5124e7c8de372226e436b330a55/regex-2026.2.28-cp310-cp310-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:3935174fa4d9f70525a4367aaff3cb8bc0548129d114260c29d9dfa4a5b41692", size = 791691, upload-time = "2026-02-28T02:16:01.752Z" }, - { url = "https://files.pythonhosted.org/packages/f5/34/96631bcf446a56ba0b2a7f684358a76855dfe315b7c2f89b35388494ede0/regex-2026.2.28-cp310-cp310-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:2b2b23587b26496ff5fd40df4278becdf386813ec00dc3533fa43a4cf0e2ad3c", size = 783111, upload-time = "2026-02-28T02:16:03.651Z" }, - { url = "https://files.pythonhosted.org/packages/39/54/f95cb7a85fe284d41cd2f3625e0f2ae30172b55dfd2af1d9b4eaef6259d7/regex-2026.2.28-cp310-cp310-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:3b24bd7e9d85dc7c6a8bd2aa14ecd234274a0248335a02adeb25448aecdd420d", size = 767512, upload-time = "2026-02-28T02:16:05.616Z" }, - { url = "https://files.pythonhosted.org/packages/3d/af/a650f64a79c02a97f73f64d4e7fc4cc1984e64affab14075e7c1f9a2db34/regex-2026.2.28-cp310-cp310-musllinux_1_2_aarch64.whl", hash = "sha256:bd477d5f79920338107f04aa645f094032d9e3030cc55be581df3d1ef61aa318", size = 773920, upload-time = "2026-02-28T02:16:08.325Z" }, - { url = "https://files.pythonhosted.org/packages/72/f8/3f9c2c2af37aedb3f5a1e7227f81bea065028785260d9cacc488e43e6997/regex-2026.2.28-cp310-cp310-musllinux_1_2_ppc64le.whl", hash = "sha256:b49eb78048c6354f49e91e4b77da21257fecb92256b6d599ae44403cab30b05b", size = 846681, upload-time = "2026-02-28T02:16:10.381Z" }, - { url = "https://files.pythonhosted.org/packages/54/12/8db04a334571359f4d127d8f89550917ec6561a2fddfd69cd91402b47482/regex-2026.2.28-cp310-cp310-musllinux_1_2_riscv64.whl", hash = "sha256:a25c7701e4f7a70021db9aaf4a4a0a67033c6318752146e03d1b94d32006217e", size = 755565, upload-time = "2026-02-28T02:16:11.972Z" }, - { url = "https://files.pythonhosted.org/packages/da/bc/91c22f384d79324121b134c267a86ca90d11f8016aafb1dc5bee05890ee3/regex-2026.2.28-cp310-cp310-musllinux_1_2_s390x.whl", hash = "sha256:9dd450db6458387167e033cfa80887a34c99c81d26da1bf8b0b41bf8c9cac88e", size = 835789, upload-time = "2026-02-28T02:16:14.036Z" }, - { url = "https://files.pythonhosted.org/packages/46/a7/4cc94fd3af01dcfdf5a9ed75c8e15fd80fcd62cc46da7592b1749e9c35db/regex-2026.2.28-cp310-cp310-musllinux_1_2_x86_64.whl", hash = "sha256:2954379dd20752e82d22accf3ff465311cbb2bac6c1f92c4afd400e1757f7451", size = 780094, upload-time = "2026-02-28T02:16:15.468Z" }, - { url = "https://files.pythonhosted.org/packages/3c/21/e5a38f420af3c77cab4a65f0c3a55ec02ac9babf04479cfd282d356988a6/regex-2026.2.28-cp310-cp310-win32.whl", hash = "sha256:1f8b17be5c27a684ea6759983c13506bd77bfc7c0347dff41b18ce5ddd2ee09a", size = 266025, upload-time = "2026-02-28T02:16:16.828Z" }, - { url = "https://files.pythonhosted.org/packages/4d/0a/205c4c1466a36e04d90afcd01d8908bac327673050c7fe316b2416d99d3d/regex-2026.2.28-cp310-cp310-win_amd64.whl", hash = "sha256:dd8847c4978bc3c7e6c826fb745f5570e518b8459ac2892151ce6627c7bc00d5", size = 277965, upload-time = "2026-02-28T02:16:18.752Z" }, - { url = "https://files.pythonhosted.org/packages/c3/4d/29b58172f954b6ec2c5ed28529a65e9026ab96b4b7016bcd3858f1c31d3c/regex-2026.2.28-cp310-cp310-win_arm64.whl", hash = "sha256:73cdcdbba8028167ea81490c7f45280113e41db2c7afb65a276f4711fa3bcbff", size = 270336, upload-time = "2026-02-28T02:16:20.735Z" }, { url = "https://files.pythonhosted.org/packages/04/db/8cbfd0ba3f302f2d09dd0019a9fcab74b63fee77a76c937d0e33161fb8c1/regex-2026.2.28-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:e621fb7c8dc147419b28e1702f58a0177ff8308a76fa295c71f3e7827849f5d9", size = 488462, upload-time = "2026-02-28T02:16:22.616Z" }, { url = "https://files.pythonhosted.org/packages/5d/10/ccc22c52802223f2368731964ddd117799e1390ffc39dbb31634a83022ee/regex-2026.2.28-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:0d5bef2031cbf38757a0b0bc4298bb4824b6332d28edc16b39247228fbdbad97", size = 290774, upload-time = "2026-02-28T02:16:23.993Z" }, { url = "https://files.pythonhosted.org/packages/62/b9/6796b3bf3101e64117201aaa3a5a030ec677ecf34b3cd6141b5d5c6c67d5/regex-2026.2.28-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:bcb399ed84eabf4282587ba151f2732ad8168e66f1d3f85b1d038868fe547703", size = 288724, upload-time = "2026-02-28T02:16:25.403Z" }, @@ -1827,20 +1667,6 @@ version = "0.30.0" source = { registry = "https://pypi.org/simple" } sdist = { url = "https://files.pythonhosted.org/packages/20/af/3f2f423103f1113b36230496629986e0ef7e199d2aa8392452b484b38ced/rpds_py-0.30.0.tar.gz", hash = "sha256:dd8ff7cf90014af0c0f787eea34794ebf6415242ee1d6fa91eaba725cc441e84", size = 69469, upload-time = "2025-11-30T20:24:38.837Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/06/0c/0c411a0ec64ccb6d104dcabe0e713e05e153a9a2c3c2bd2b32ce412166fe/rpds_py-0.30.0-cp310-cp310-macosx_10_12_x86_64.whl", hash = "sha256:679ae98e00c0e8d68a7fda324e16b90fd5260945b45d3b824c892cec9eea3288", size = 370490, upload-time = "2025-11-30T20:21:33.256Z" }, - { url = "https://files.pythonhosted.org/packages/19/6a/4ba3d0fb7297ebae71171822554abe48d7cab29c28b8f9f2c04b79988c05/rpds_py-0.30.0-cp310-cp310-macosx_11_0_arm64.whl", hash = "sha256:4cc2206b76b4f576934f0ed374b10d7ca5f457858b157ca52064bdfc26b9fc00", size = 359751, upload-time = "2025-11-30T20:21:34.591Z" }, - { url = "https://files.pythonhosted.org/packages/cd/7c/e4933565ef7f7a0818985d87c15d9d273f1a649afa6a52ea35ad011195ea/rpds_py-0.30.0-cp310-cp310-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:389a2d49eded1896c3d48b0136ead37c48e221b391c052fba3f4055c367f60a6", size = 389696, upload-time = "2025-11-30T20:21:36.122Z" }, - { url = "https://files.pythonhosted.org/packages/5e/01/6271a2511ad0815f00f7ed4390cf2567bec1d4b1da39e2c27a41e6e3b4de/rpds_py-0.30.0-cp310-cp310-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:32c8528634e1bf7121f3de08fa85b138f4e0dc47657866630611b03967f041d7", size = 403136, upload-time = "2025-11-30T20:21:37.728Z" }, - { url = "https://files.pythonhosted.org/packages/55/64/c857eb7cd7541e9b4eee9d49c196e833128a55b89a9850a9c9ac33ccf897/rpds_py-0.30.0-cp310-cp310-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:f207f69853edd6f6700b86efb84999651baf3789e78a466431df1331608e5324", size = 524699, upload-time = "2025-11-30T20:21:38.92Z" }, - { url = "https://files.pythonhosted.org/packages/9c/ed/94816543404078af9ab26159c44f9e98e20fe47e2126d5d32c9d9948d10a/rpds_py-0.30.0-cp310-cp310-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:67b02ec25ba7a9e8fa74c63b6ca44cf5707f2fbfadae3ee8e7494297d56aa9df", size = 412022, upload-time = "2025-11-30T20:21:40.407Z" }, - { url = "https://files.pythonhosted.org/packages/61/b5/707f6cf0066a6412aacc11d17920ea2e19e5b2f04081c64526eb35b5c6e7/rpds_py-0.30.0-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:0c0e95f6819a19965ff420f65578bacb0b00f251fefe2c8b23347c37174271f3", size = 390522, upload-time = "2025-11-30T20:21:42.17Z" }, - { url = "https://files.pythonhosted.org/packages/13/4e/57a85fda37a229ff4226f8cbcf09f2a455d1ed20e802ce5b2b4a7f5ed053/rpds_py-0.30.0-cp310-cp310-manylinux_2_31_riscv64.whl", hash = "sha256:a452763cc5198f2f98898eb98f7569649fe5da666c2dc6b5ddb10fde5a574221", size = 404579, upload-time = "2025-11-30T20:21:43.769Z" }, - { url = "https://files.pythonhosted.org/packages/f9/da/c9339293513ec680a721e0e16bf2bac3db6e5d7e922488de471308349bba/rpds_py-0.30.0-cp310-cp310-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:e0b65193a413ccc930671c55153a03ee57cecb49e6227204b04fae512eb657a7", size = 421305, upload-time = "2025-11-30T20:21:44.994Z" }, - { url = "https://files.pythonhosted.org/packages/f9/be/522cb84751114f4ad9d822ff5a1aa3c98006341895d5f084779b99596e5c/rpds_py-0.30.0-cp310-cp310-musllinux_1_2_aarch64.whl", hash = "sha256:858738e9c32147f78b3ac24dc0edb6610000e56dc0f700fd5f651d0a0f0eb9ff", size = 572503, upload-time = "2025-11-30T20:21:46.91Z" }, - { url = "https://files.pythonhosted.org/packages/a2/9b/de879f7e7ceddc973ea6e4629e9b380213a6938a249e94b0cdbcc325bb66/rpds_py-0.30.0-cp310-cp310-musllinux_1_2_i686.whl", hash = "sha256:da279aa314f00acbb803da1e76fa18666778e8a8f83484fba94526da5de2cba7", size = 598322, upload-time = "2025-11-30T20:21:48.709Z" }, - { url = "https://files.pythonhosted.org/packages/48/ac/f01fc22efec3f37d8a914fc1b2fb9bcafd56a299edbe96406f3053edea5a/rpds_py-0.30.0-cp310-cp310-musllinux_1_2_x86_64.whl", hash = "sha256:7c64d38fb49b6cdeda16ab49e35fe0da2e1e9b34bc38bd78386530f218b37139", size = 560792, upload-time = "2025-11-30T20:21:50.024Z" }, - { url = "https://files.pythonhosted.org/packages/e2/da/4e2b19d0f131f35b6146425f846563d0ce036763e38913d917187307a671/rpds_py-0.30.0-cp310-cp310-win32.whl", hash = "sha256:6de2a32a1665b93233cde140ff8b3467bdb9e2af2b91079f0333a0974d12d464", size = 221901, upload-time = "2025-11-30T20:21:51.32Z" }, - { url = "https://files.pythonhosted.org/packages/96/cb/156d7a5cf4f78a7cc571465d8aec7a3c447c94f6749c5123f08438bcf7bc/rpds_py-0.30.0-cp310-cp310-win_amd64.whl", hash = "sha256:1726859cd0de969f88dc8673bdd954185b9104e05806be64bcd87badbe313169", size = 235823, upload-time = "2025-11-30T20:21:52.505Z" }, { url = "https://files.pythonhosted.org/packages/4d/6e/f964e88b3d2abee2a82c1ac8366da848fce1c6d834dc2132c3fda3970290/rpds_py-0.30.0-cp311-cp311-macosx_10_12_x86_64.whl", hash = "sha256:a2bffea6a4ca9f01b3f8e548302470306689684e61602aa3d141e34da06cf425", size = 370157, upload-time = "2025-11-30T20:21:53.789Z" }, { url = "https://files.pythonhosted.org/packages/94/ba/24e5ebb7c1c82e74c4e4f33b2112a5573ddc703915b13a073737b59b86e0/rpds_py-0.30.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:dc4f992dfe1e2bc3ebc7444f6c7051b4bc13cd8e33e43511e8ffd13bf407010d", size = 359676, upload-time = "2025-11-30T20:21:55.475Z" }, { url = "https://files.pythonhosted.org/packages/84/86/04dbba1b087227747d64d80c3b74df946b986c57af0a9f0c98726d4d7a3b/rpds_py-0.30.0-cp311-cp311-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:422c3cb9856d80b09d30d2eb255d0754b23e090034e1deb4083f8004bd0761e4", size = 389938, upload-time = "2025-11-30T20:21:57.079Z" }, @@ -1996,13 +1822,6 @@ dependencies = [ ] sdist = { url = "https://files.pythonhosted.org/packages/7d/ab/4d017d0f76ec3171d469d80fc03dfbb4e48a4bcaddaa831b31d526f05edc/tiktoken-0.12.0.tar.gz", hash = "sha256:b18ba7ee2b093863978fcb14f74b3707cdc8d4d4d3836853ce7ec60772139931", size = 37806, upload-time = "2025-10-06T20:22:45.419Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/89/b3/2cb7c17b6c4cf8ca983204255d3f1d95eda7213e247e6947a0ee2c747a2c/tiktoken-0.12.0-cp310-cp310-macosx_10_12_x86_64.whl", hash = "sha256:3de02f5a491cfd179aec916eddb70331814bd6bf764075d39e21d5862e533970", size = 1051991, upload-time = "2025-10-06T20:21:34.098Z" }, - { url = "https://files.pythonhosted.org/packages/27/0f/df139f1df5f6167194ee5ab24634582ba9a1b62c6b996472b0277ec80f66/tiktoken-0.12.0-cp310-cp310-macosx_11_0_arm64.whl", hash = "sha256:b6cfb6d9b7b54d20af21a912bfe63a2727d9cfa8fbda642fd8322c70340aad16", size = 995798, upload-time = "2025-10-06T20:21:35.579Z" }, - { url = "https://files.pythonhosted.org/packages/ef/5d/26a691f28ab220d5edc09b9b787399b130f24327ef824de15e5d85ef21aa/tiktoken-0.12.0-cp310-cp310-manylinux_2_28_aarch64.whl", hash = "sha256:cde24cdb1b8a08368f709124f15b36ab5524aac5fa830cc3fdce9c03d4fb8030", size = 1129865, upload-time = "2025-10-06T20:21:36.675Z" }, - { url = "https://files.pythonhosted.org/packages/b2/94/443fab3d4e5ebecac895712abd3849b8da93b7b7dec61c7db5c9c7ebe40c/tiktoken-0.12.0-cp310-cp310-manylinux_2_28_x86_64.whl", hash = "sha256:6de0da39f605992649b9cfa6f84071e3f9ef2cec458d08c5feb1b6f0ff62e134", size = 1152856, upload-time = "2025-10-06T20:21:37.873Z" }, - { url = "https://files.pythonhosted.org/packages/54/35/388f941251b2521c70dd4c5958e598ea6d2c88e28445d2fb8189eecc1dfc/tiktoken-0.12.0-cp310-cp310-musllinux_1_2_aarch64.whl", hash = "sha256:6faa0534e0eefbcafaccb75927a4a380463a2eaa7e26000f0173b920e98b720a", size = 1195308, upload-time = "2025-10-06T20:21:39.577Z" }, - { url = "https://files.pythonhosted.org/packages/f8/00/c6681c7f833dd410576183715a530437a9873fa910265817081f65f9105f/tiktoken-0.12.0-cp310-cp310-musllinux_1_2_x86_64.whl", hash = "sha256:82991e04fc860afb933efb63957affc7ad54f83e2216fe7d319007dab1ba5892", size = 1255697, upload-time = "2025-10-06T20:21:41.154Z" }, - { url = "https://files.pythonhosted.org/packages/5f/d2/82e795a6a9bafa034bf26a58e68fe9a89eeaaa610d51dbeb22106ba04f0a/tiktoken-0.12.0-cp310-cp310-win_amd64.whl", hash = "sha256:6fb2995b487c2e31acf0a9e17647e3b242235a20832642bb7a9d1a181c0c1bb1", size = 879375, upload-time = "2025-10-06T20:21:43.201Z" }, { url = "https://files.pythonhosted.org/packages/de/46/21ea696b21f1d6d1efec8639c204bdf20fde8bafb351e1355c72c5d7de52/tiktoken-0.12.0-cp311-cp311-macosx_10_12_x86_64.whl", hash = "sha256:6e227c7f96925003487c33b1b32265fad2fbcec2b7cf4817afb76d416f40f6bb", size = 1051565, upload-time = "2025-10-06T20:21:44.566Z" }, { url = "https://files.pythonhosted.org/packages/c9/d9/35c5d2d9e22bb2a5f74ba48266fb56c63d76ae6f66e02feb628671c0283e/tiktoken-0.12.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:c06cf0fcc24c2cb2adb5e185c7082a82cba29c17575e828518c2f11a01f445aa", size = 995284, upload-time = "2025-10-06T20:21:45.622Z" }, { url = "https://files.pythonhosted.org/packages/01/84/961106c37b8e49b9fdcf33fe007bb3a8fdcc380c528b20cc7fbba80578b8/tiktoken-0.12.0-cp311-cp311-manylinux_2_28_aarch64.whl", hash = "sha256:f18f249b041851954217e9fd8e5c00b024ab2315ffda5ed77665a05fa91f42dc", size = 1129201, upload-time = "2025-10-06T20:21:47.074Z" }, @@ -2047,60 +1866,6 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/af/df/c7891ef9d2712ad774777271d39fdef63941ffba0a9d59b7ad1fd2765e57/tiktoken-0.12.0-cp314-cp314t-win_amd64.whl", hash = "sha256:f61c0aea5565ac82e2ec50a05e02a6c44734e91b51c10510b084ea1b8e633a71", size = 920667, upload-time = "2025-10-06T20:22:34.444Z" }, ] -[[package]] -name = "tomli" -version = "2.4.0" -source = { registry = "https://pypi.org/simple" } -sdist = { url = "https://files.pythonhosted.org/packages/82/30/31573e9457673ab10aa432461bee537ce6cef177667deca369efb79df071/tomli-2.4.0.tar.gz", hash = "sha256:aa89c3f6c277dd275d8e243ad24f3b5e701491a860d5121f2cdd399fbb31fc9c", size = 17477, upload-time = "2026-01-11T11:22:38.165Z" } -wheels = [ - { url = "https://files.pythonhosted.org/packages/3c/d9/3dc2289e1f3b32eb19b9785b6a006b28ee99acb37d1d47f78d4c10e28bf8/tomli-2.4.0-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:b5ef256a3fd497d4973c11bf142e9ed78b150d36f5773f1ca6088c230ffc5867", size = 153663, upload-time = "2026-01-11T11:21:45.27Z" }, - { url = "https://files.pythonhosted.org/packages/51/32/ef9f6845e6b9ca392cd3f64f9ec185cc6f09f0a2df3db08cbe8809d1d435/tomli-2.4.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:5572e41282d5268eb09a697c89a7bee84fae66511f87533a6f88bd2f7b652da9", size = 148469, upload-time = "2026-01-11T11:21:46.873Z" }, - { url = "https://files.pythonhosted.org/packages/d6/c2/506e44cce89a8b1b1e047d64bd495c22c9f71f21e05f380f1a950dd9c217/tomli-2.4.0-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:551e321c6ba03b55676970b47cb1b73f14a0a4dce6a3e1a9458fd6d921d72e95", size = 236039, upload-time = "2026-01-11T11:21:48.503Z" }, - { url = "https://files.pythonhosted.org/packages/b3/40/e1b65986dbc861b7e986e8ec394598187fa8aee85b1650b01dd925ca0be8/tomli-2.4.0-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:5e3f639a7a8f10069d0e15408c0b96a2a828cfdec6fca05296ebcdcc28ca7c76", size = 243007, upload-time = "2026-01-11T11:21:49.456Z" }, - { url = "https://files.pythonhosted.org/packages/9c/6f/6e39ce66b58a5b7ae572a0f4352ff40c71e8573633deda43f6a379d56b3e/tomli-2.4.0-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:1b168f2731796b045128c45982d3a4874057626da0e2ef1fdd722848b741361d", size = 240875, upload-time = "2026-01-11T11:21:50.755Z" }, - { url = "https://files.pythonhosted.org/packages/aa/ad/cb089cb190487caa80204d503c7fd0f4d443f90b95cf4ef5cf5aa0f439b0/tomli-2.4.0-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:133e93646ec4300d651839d382d63edff11d8978be23da4cc106f5a18b7d0576", size = 246271, upload-time = "2026-01-11T11:21:51.81Z" }, - { url = "https://files.pythonhosted.org/packages/0b/63/69125220e47fd7a3a27fd0de0c6398c89432fec41bc739823bcc66506af6/tomli-2.4.0-cp311-cp311-win32.whl", hash = "sha256:b6c78bdf37764092d369722d9946cb65b8767bfa4110f902a1b2542d8d173c8a", size = 96770, upload-time = "2026-01-11T11:21:52.647Z" }, - { url = "https://files.pythonhosted.org/packages/1e/0d/a22bb6c83f83386b0008425a6cd1fa1c14b5f3dd4bad05e98cf3dbbf4a64/tomli-2.4.0-cp311-cp311-win_amd64.whl", hash = "sha256:d3d1654e11d724760cdb37a3d7691f0be9db5fbdaef59c9f532aabf87006dbaa", size = 107626, upload-time = "2026-01-11T11:21:53.459Z" }, - { url = "https://files.pythonhosted.org/packages/2f/6d/77be674a3485e75cacbf2ddba2b146911477bd887dda9d8c9dfb2f15e871/tomli-2.4.0-cp311-cp311-win_arm64.whl", hash = "sha256:cae9c19ed12d4e8f3ebf46d1a75090e4c0dc16271c5bce1c833ac168f08fb614", size = 94842, upload-time = "2026-01-11T11:21:54.831Z" }, - { url = "https://files.pythonhosted.org/packages/3c/43/7389a1869f2f26dba52404e1ef13b4784b6b37dac93bac53457e3ff24ca3/tomli-2.4.0-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:920b1de295e72887bafa3ad9f7a792f811847d57ea6b1215154030cf131f16b1", size = 154894, upload-time = "2026-01-11T11:21:56.07Z" }, - { url = "https://files.pythonhosted.org/packages/e9/05/2f9bf110b5294132b2edf13fe6ca6ae456204f3d749f623307cbb7a946f2/tomli-2.4.0-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:7d6d9a4aee98fac3eab4952ad1d73aee87359452d1c086b5ceb43ed02ddb16b8", size = 149053, upload-time = "2026-01-11T11:21:57.467Z" }, - { url = "https://files.pythonhosted.org/packages/e8/41/1eda3ca1abc6f6154a8db4d714a4d35c4ad90adc0bcf700657291593fbf3/tomli-2.4.0-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:36b9d05b51e65b254ea6c2585b59d2c4cb91c8a3d91d0ed0f17591a29aaea54a", size = 243481, upload-time = "2026-01-11T11:21:58.661Z" }, - { url = "https://files.pythonhosted.org/packages/d2/6d/02ff5ab6c8868b41e7d4b987ce2b5f6a51d3335a70aa144edd999e055a01/tomli-2.4.0-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:1c8a885b370751837c029ef9bc014f27d80840e48bac415f3412e6593bbc18c1", size = 251720, upload-time = "2026-01-11T11:22:00.178Z" }, - { url = "https://files.pythonhosted.org/packages/7b/57/0405c59a909c45d5b6f146107c6d997825aa87568b042042f7a9c0afed34/tomli-2.4.0-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:8768715ffc41f0008abe25d808c20c3d990f42b6e2e58305d5da280ae7d1fa3b", size = 247014, upload-time = "2026-01-11T11:22:01.238Z" }, - { url = "https://files.pythonhosted.org/packages/2c/0e/2e37568edd944b4165735687cbaf2fe3648129e440c26d02223672ee0630/tomli-2.4.0-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:7b438885858efd5be02a9a133caf5812b8776ee0c969fea02c45e8e3f296ba51", size = 251820, upload-time = "2026-01-11T11:22:02.727Z" }, - { url = "https://files.pythonhosted.org/packages/5a/1c/ee3b707fdac82aeeb92d1a113f803cf6d0f37bdca0849cb489553e1f417a/tomli-2.4.0-cp312-cp312-win32.whl", hash = "sha256:0408e3de5ec77cc7f81960c362543cbbd91ef883e3138e81b729fc3eea5b9729", size = 97712, upload-time = "2026-01-11T11:22:03.777Z" }, - { url = "https://files.pythonhosted.org/packages/69/13/c07a9177d0b3bab7913299b9278845fc6eaaca14a02667c6be0b0a2270c8/tomli-2.4.0-cp312-cp312-win_amd64.whl", hash = "sha256:685306e2cc7da35be4ee914fd34ab801a6acacb061b6a7abca922aaf9ad368da", size = 108296, upload-time = "2026-01-11T11:22:04.86Z" }, - { url = "https://files.pythonhosted.org/packages/18/27/e267a60bbeeee343bcc279bb9e8fbed0cbe224bc7b2a3dc2975f22809a09/tomli-2.4.0-cp312-cp312-win_arm64.whl", hash = "sha256:5aa48d7c2356055feef06a43611fc401a07337d5b006be13a30f6c58f869e3c3", size = 94553, upload-time = "2026-01-11T11:22:05.854Z" }, - { url = "https://files.pythonhosted.org/packages/34/91/7f65f9809f2936e1f4ce6268ae1903074563603b2a2bd969ebbda802744f/tomli-2.4.0-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:84d081fbc252d1b6a982e1870660e7330fb8f90f676f6e78b052ad4e64714bf0", size = 154915, upload-time = "2026-01-11T11:22:06.703Z" }, - { url = "https://files.pythonhosted.org/packages/20/aa/64dd73a5a849c2e8f216b755599c511badde80e91e9bc2271baa7b2cdbb1/tomli-2.4.0-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:9a08144fa4cba33db5255f9b74f0b89888622109bd2776148f2597447f92a94e", size = 149038, upload-time = "2026-01-11T11:22:07.56Z" }, - { url = "https://files.pythonhosted.org/packages/9e/8a/6d38870bd3d52c8d1505ce054469a73f73a0fe62c0eaf5dddf61447e32fa/tomli-2.4.0-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:c73add4bb52a206fd0c0723432db123c0c75c280cbd67174dd9d2db228ebb1b4", size = 242245, upload-time = "2026-01-11T11:22:08.344Z" }, - { url = "https://files.pythonhosted.org/packages/59/bb/8002fadefb64ab2669e5b977df3f5e444febea60e717e755b38bb7c41029/tomli-2.4.0-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:1fb2945cbe303b1419e2706e711b7113da57b7db31ee378d08712d678a34e51e", size = 250335, upload-time = "2026-01-11T11:22:09.951Z" }, - { url = "https://files.pythonhosted.org/packages/a5/3d/4cdb6f791682b2ea916af2de96121b3cb1284d7c203d97d92d6003e91c8d/tomli-2.4.0-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:bbb1b10aa643d973366dc2cb1ad94f99c1726a02343d43cbc011edbfac579e7c", size = 245962, upload-time = "2026-01-11T11:22:11.27Z" }, - { url = "https://files.pythonhosted.org/packages/f2/4a/5f25789f9a460bd858ba9756ff52d0830d825b458e13f754952dd15fb7bb/tomli-2.4.0-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:4cbcb367d44a1f0c2be408758b43e1ffb5308abe0ea222897d6bfc8e8281ef2f", size = 250396, upload-time = "2026-01-11T11:22:12.325Z" }, - { url = "https://files.pythonhosted.org/packages/aa/2f/b73a36fea58dfa08e8b3a268750e6853a6aac2a349241a905ebd86f3047a/tomli-2.4.0-cp313-cp313-win32.whl", hash = "sha256:7d49c66a7d5e56ac959cb6fc583aff0651094ec071ba9ad43df785abc2320d86", size = 97530, upload-time = "2026-01-11T11:22:13.865Z" }, - { url = "https://files.pythonhosted.org/packages/3b/af/ca18c134b5d75de7e8dc551c5234eaba2e8e951f6b30139599b53de9c187/tomli-2.4.0-cp313-cp313-win_amd64.whl", hash = "sha256:3cf226acb51d8f1c394c1b310e0e0e61fecdd7adcb78d01e294ac297dd2e7f87", size = 108227, upload-time = "2026-01-11T11:22:15.224Z" }, - { url = "https://files.pythonhosted.org/packages/22/c3/b386b832f209fee8073c8138ec50f27b4460db2fdae9ffe022df89a57f9b/tomli-2.4.0-cp313-cp313-win_arm64.whl", hash = "sha256:d20b797a5c1ad80c516e41bc1fb0443ddb5006e9aaa7bda2d71978346aeb9132", size = 94748, upload-time = "2026-01-11T11:22:16.009Z" }, - { url = "https://files.pythonhosted.org/packages/f3/c4/84047a97eb1004418bc10bdbcfebda209fca6338002eba2dc27cc6d13563/tomli-2.4.0-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:26ab906a1eb794cd4e103691daa23d95c6919cc2fa9160000ac02370cc9dd3f6", size = 154725, upload-time = "2026-01-11T11:22:17.269Z" }, - { url = "https://files.pythonhosted.org/packages/a8/5d/d39038e646060b9d76274078cddf146ced86dc2b9e8bbf737ad5983609a0/tomli-2.4.0-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:20cedb4ee43278bc4f2fee6cb50daec836959aadaf948db5172e776dd3d993fc", size = 148901, upload-time = "2026-01-11T11:22:18.287Z" }, - { url = "https://files.pythonhosted.org/packages/73/e5/383be1724cb30f4ce44983d249645684a48c435e1cd4f8b5cded8a816d3c/tomli-2.4.0-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:39b0b5d1b6dd03684b3fb276407ebed7090bbec989fa55838c98560c01113b66", size = 243375, upload-time = "2026-01-11T11:22:19.154Z" }, - { url = "https://files.pythonhosted.org/packages/31/f0/bea80c17971c8d16d3cc109dc3585b0f2ce1036b5f4a8a183789023574f2/tomli-2.4.0-cp314-cp314-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:a26d7ff68dfdb9f87a016ecfd1e1c2bacbe3108f4e0f8bcd2228ef9a766c787d", size = 250639, upload-time = "2026-01-11T11:22:20.168Z" }, - { url = "https://files.pythonhosted.org/packages/2c/8f/2853c36abbb7608e3f945d8a74e32ed3a74ee3a1f468f1ffc7d1cb3abba6/tomli-2.4.0-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:20ffd184fb1df76a66e34bd1b36b4a4641bd2b82954befa32fe8163e79f1a702", size = 246897, upload-time = "2026-01-11T11:22:21.544Z" }, - { url = "https://files.pythonhosted.org/packages/49/f0/6c05e3196ed5337b9fe7ea003e95fd3819a840b7a0f2bf5a408ef1dad8ed/tomli-2.4.0-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:75c2f8bbddf170e8effc98f5e9084a8751f8174ea6ccf4fca5398436e0320bc8", size = 254697, upload-time = "2026-01-11T11:22:23.058Z" }, - { url = "https://files.pythonhosted.org/packages/f3/f5/2922ef29c9f2951883525def7429967fc4d8208494e5ab524234f06b688b/tomli-2.4.0-cp314-cp314-win32.whl", hash = "sha256:31d556d079d72db7c584c0627ff3a24c5d3fb4f730221d3444f3efb1b2514776", size = 98567, upload-time = "2026-01-11T11:22:24.033Z" }, - { url = "https://files.pythonhosted.org/packages/7b/31/22b52e2e06dd2a5fdbc3ee73226d763b184ff21fc24e20316a44ccc4d96b/tomli-2.4.0-cp314-cp314-win_amd64.whl", hash = "sha256:43e685b9b2341681907759cf3a04e14d7104b3580f808cfde1dfdb60ada85475", size = 108556, upload-time = "2026-01-11T11:22:25.378Z" }, - { url = "https://files.pythonhosted.org/packages/48/3d/5058dff3255a3d01b705413f64f4306a141a8fd7a251e5a495e3f192a998/tomli-2.4.0-cp314-cp314-win_arm64.whl", hash = "sha256:3d895d56bd3f82ddd6faaff993c275efc2ff38e52322ea264122d72729dca2b2", size = 96014, upload-time = "2026-01-11T11:22:26.138Z" }, - { url = "https://files.pythonhosted.org/packages/b8/4e/75dab8586e268424202d3a1997ef6014919c941b50642a1682df43204c22/tomli-2.4.0-cp314-cp314t-macosx_10_15_x86_64.whl", hash = "sha256:5b5807f3999fb66776dbce568cc9a828544244a8eb84b84b9bafc080c99597b9", size = 163339, upload-time = "2026-01-11T11:22:27.143Z" }, - { url = "https://files.pythonhosted.org/packages/06/e3/b904d9ab1016829a776d97f163f183a48be6a4deb87304d1e0116a349519/tomli-2.4.0-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:c084ad935abe686bd9c898e62a02a19abfc9760b5a79bc29644463eaf2840cb0", size = 159490, upload-time = "2026-01-11T11:22:28.399Z" }, - { url = "https://files.pythonhosted.org/packages/e3/5a/fc3622c8b1ad823e8ea98a35e3c632ee316d48f66f80f9708ceb4f2a0322/tomli-2.4.0-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:0f2e3955efea4d1cfbcb87bc321e00dc08d2bcb737fd1d5e398af111d86db5df", size = 269398, upload-time = "2026-01-11T11:22:29.345Z" }, - { url = "https://files.pythonhosted.org/packages/fd/33/62bd6152c8bdd4c305ad9faca48f51d3acb2df1f8791b1477d46ff86e7f8/tomli-2.4.0-cp314-cp314t-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:0e0fe8a0b8312acf3a88077a0802565cb09ee34107813bba1c7cd591fa6cfc8d", size = 276515, upload-time = "2026-01-11T11:22:30.327Z" }, - { url = "https://files.pythonhosted.org/packages/4b/ff/ae53619499f5235ee4211e62a8d7982ba9e439a0fb4f2f351a93d67c1dd2/tomli-2.4.0-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:413540dce94673591859c4c6f794dfeaa845e98bf35d72ed59636f869ef9f86f", size = 273806, upload-time = "2026-01-11T11:22:32.56Z" }, - { url = "https://files.pythonhosted.org/packages/47/71/cbca7787fa68d4d0a9f7072821980b39fbb1b6faeb5f5cf02f4a5559fa28/tomli-2.4.0-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:0dc56fef0e2c1c470aeac5b6ca8cc7b640bb93e92d9803ddaf9ea03e198f5b0b", size = 281340, upload-time = "2026-01-11T11:22:33.505Z" }, - { url = "https://files.pythonhosted.org/packages/f5/00/d595c120963ad42474cf6ee7771ad0d0e8a49d0f01e29576ee9195d9ecdf/tomli-2.4.0-cp314-cp314t-win32.whl", hash = "sha256:d878f2a6707cc9d53a1be1414bbb419e629c3d6e67f69230217bb663e76b5087", size = 108106, upload-time = "2026-01-11T11:22:34.451Z" }, - { url = "https://files.pythonhosted.org/packages/de/69/9aa0c6a505c2f80e519b43764f8b4ba93b5a0bbd2d9a9de6e2b24271b9a5/tomli-2.4.0-cp314-cp314t-win_amd64.whl", hash = "sha256:2add28aacc7425117ff6364fe9e06a183bb0251b03f986df0e78e974047571fd", size = 120504, upload-time = "2026-01-11T11:22:35.764Z" }, - { url = "https://files.pythonhosted.org/packages/b3/9f/f1668c281c58cfae01482f7114a4b88d345e4c140386241a1a24dcc9e7bc/tomli-2.4.0-cp314-cp314t-win_arm64.whl", hash = "sha256:2b1e3b80e1d5e52e40e9b924ec43d81570f0e7d09d11081b797bc4692765a3d4", size = 99561, upload-time = "2026-01-11T11:22:36.624Z" }, - { url = "https://files.pythonhosted.org/packages/23/d1/136eb2cb77520a31e1f64cbae9d33ec6df0d78bdf4160398e86eec8a8754/tomli-2.4.0-py3-none-any.whl", hash = "sha256:1f776e7d669ebceb01dee46484485f43a4048746235e683bcdffacdf1fb4785a", size = 14477, upload-time = "2026-01-11T11:22:37.446Z" }, -] - [[package]] name = "tqdm" version = "4.67.3" @@ -2181,7 +1946,6 @@ dependencies = [ { name = "filelock" }, { name = "platformdirs" }, { name = "python-discovery" }, - { name = "typing-extensions", marker = "python_full_version < '3.11'" }, ] sdist = { url = "https://files.pythonhosted.org/packages/aa/92/58199fe10049f9703c2666e809c4f686c54ef0a68b0f6afccf518c0b1eb9/virtualenv-21.2.0.tar.gz", hash = "sha256:1720dc3a62ef5b443092e3f499228599045d7fea4c79199770499df8becf9098", size = 5840618, upload-time = "2026-03-09T17:24:38.013Z" } wheels = [ @@ -2206,16 +1970,6 @@ version = "1.17.3" source = { registry = "https://pypi.org/simple" } sdist = { url = "https://files.pythonhosted.org/packages/95/8f/aeb76c5b46e273670962298c23e7ddde79916cb74db802131d49a85e4b7d/wrapt-1.17.3.tar.gz", hash = "sha256:f66eb08feaa410fe4eebd17f2a2c8e2e46d3476e9f8c783daa8e09e0faa666d0", size = 55547, upload-time = "2025-08-12T05:53:21.714Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/3f/23/bb82321b86411eb51e5a5db3fb8f8032fd30bd7c2d74bfe936136b2fa1d6/wrapt-1.17.3-cp310-cp310-macosx_10_9_universal2.whl", hash = "sha256:88bbae4d40d5a46142e70d58bf664a89b6b4befaea7b2ecc14e03cedb8e06c04", size = 53482, upload-time = "2025-08-12T05:51:44.467Z" }, - { url = "https://files.pythonhosted.org/packages/45/69/f3c47642b79485a30a59c63f6d739ed779fb4cc8323205d047d741d55220/wrapt-1.17.3-cp310-cp310-macosx_10_9_x86_64.whl", hash = "sha256:e6b13af258d6a9ad602d57d889f83b9d5543acd471eee12eb51f5b01f8eb1bc2", size = 38676, upload-time = "2025-08-12T05:51:32.636Z" }, - { url = "https://files.pythonhosted.org/packages/d1/71/e7e7f5670c1eafd9e990438e69d8fb46fa91a50785332e06b560c869454f/wrapt-1.17.3-cp310-cp310-macosx_11_0_arm64.whl", hash = "sha256:fd341868a4b6714a5962c1af0bd44f7c404ef78720c7de4892901e540417111c", size = 38957, upload-time = "2025-08-12T05:51:54.655Z" }, - { url = "https://files.pythonhosted.org/packages/de/17/9f8f86755c191d6779d7ddead1a53c7a8aa18bccb7cea8e7e72dfa6a8a09/wrapt-1.17.3-cp310-cp310-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:f9b2601381be482f70e5d1051a5965c25fb3625455a2bf520b5a077b22afb775", size = 81975, upload-time = "2025-08-12T05:52:30.109Z" }, - { url = "https://files.pythonhosted.org/packages/f2/15/dd576273491f9f43dd09fce517f6c2ce6eb4fe21681726068db0d0467096/wrapt-1.17.3-cp310-cp310-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:343e44b2a8e60e06a7e0d29c1671a0d9951f59174f3709962b5143f60a2a98bd", size = 83149, upload-time = "2025-08-12T05:52:09.316Z" }, - { url = "https://files.pythonhosted.org/packages/0c/c4/5eb4ce0d4814521fee7aa806264bf7a114e748ad05110441cd5b8a5c744b/wrapt-1.17.3-cp310-cp310-musllinux_1_2_aarch64.whl", hash = "sha256:33486899acd2d7d3066156b03465b949da3fd41a5da6e394ec49d271baefcf05", size = 82209, upload-time = "2025-08-12T05:52:10.331Z" }, - { url = "https://files.pythonhosted.org/packages/31/4b/819e9e0eb5c8dc86f60dfc42aa4e2c0d6c3db8732bce93cc752e604bb5f5/wrapt-1.17.3-cp310-cp310-musllinux_1_2_x86_64.whl", hash = "sha256:e6f40a8aa5a92f150bdb3e1c44b7e98fb7113955b2e5394122fa5532fec4b418", size = 81551, upload-time = "2025-08-12T05:52:31.137Z" }, - { url = "https://files.pythonhosted.org/packages/f8/83/ed6baf89ba3a56694700139698cf703aac9f0f9eb03dab92f57551bd5385/wrapt-1.17.3-cp310-cp310-win32.whl", hash = "sha256:a36692b8491d30a8c75f1dfee65bef119d6f39ea84ee04d9f9311f83c5ad9390", size = 36464, upload-time = "2025-08-12T05:53:01.204Z" }, - { url = "https://files.pythonhosted.org/packages/2f/90/ee61d36862340ad7e9d15a02529df6b948676b9a5829fd5e16640156627d/wrapt-1.17.3-cp310-cp310-win_amd64.whl", hash = "sha256:afd964fd43b10c12213574db492cb8f73b2f0826c8df07a68288f8f19af2ebe6", size = 38748, upload-time = "2025-08-12T05:53:00.209Z" }, - { url = "https://files.pythonhosted.org/packages/bd/c3/cefe0bd330d389c9983ced15d326f45373f4073c9f4a8c2f99b50bfea329/wrapt-1.17.3-cp310-cp310-win_arm64.whl", hash = "sha256:af338aa93554be859173c39c85243970dc6a289fa907402289eeae7543e1ae18", size = 36810, upload-time = "2025-08-12T05:52:51.906Z" }, { url = "https://files.pythonhosted.org/packages/52/db/00e2a219213856074a213503fdac0511203dceefff26e1daa15250cc01a0/wrapt-1.17.3-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:273a736c4645e63ac582c60a56b0acb529ef07f78e08dc6bfadf6a46b19c0da7", size = 53482, upload-time = "2025-08-12T05:51:45.79Z" }, { url = "https://files.pythonhosted.org/packages/5e/30/ca3c4a5eba478408572096fe9ce36e6e915994dd26a4e9e98b4f729c06d9/wrapt-1.17.3-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:5531d911795e3f935a9c23eb1c8c03c211661a5060aab167065896bbf62a5f85", size = 38674, upload-time = "2025-08-12T05:51:34.629Z" }, { url = "https://files.pythonhosted.org/packages/31/25/3e8cc2c46b5329c5957cec959cb76a10718e1a513309c31399a4dad07eb3/wrapt-1.17.3-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:0610b46293c59a3adbae3dee552b648b984176f8562ee0dba099a56cfbe4df1f", size = 38959, upload-time = "2025-08-12T05:51:56.074Z" }, @@ -2275,21 +2029,6 @@ version = "3.6.0" source = { registry = "https://pypi.org/simple" } sdist = { url = "https://files.pythonhosted.org/packages/02/84/30869e01909fb37a6cc7e18688ee8bf1e42d57e7e0777636bd47524c43c7/xxhash-3.6.0.tar.gz", hash = "sha256:f0162a78b13a0d7617b2845b90c763339d1f1d82bb04a4b07f4ab535cc5e05d6", size = 85160, upload-time = "2025-10-02T14:37:08.097Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/34/ee/f9f1d656ad168681bb0f6b092372c1e533c4416b8069b1896a175c46e484/xxhash-3.6.0-cp310-cp310-macosx_10_9_x86_64.whl", hash = "sha256:87ff03d7e35c61435976554477a7f4cd1704c3596a89a8300d5ce7fc83874a71", size = 32845, upload-time = "2025-10-02T14:33:51.573Z" }, - { url = "https://files.pythonhosted.org/packages/a3/b1/93508d9460b292c74a09b83d16750c52a0ead89c51eea9951cb97a60d959/xxhash-3.6.0-cp310-cp310-macosx_11_0_arm64.whl", hash = "sha256:f572dfd3d0e2eb1a57511831cf6341242f5a9f8298a45862d085f5b93394a27d", size = 30807, upload-time = "2025-10-02T14:33:52.964Z" }, - { url = "https://files.pythonhosted.org/packages/07/55/28c93a3662f2d200c70704efe74aab9640e824f8ce330d8d3943bf7c9b3c/xxhash-3.6.0-cp310-cp310-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", hash = "sha256:89952ea539566b9fed2bbd94e589672794b4286f342254fad28b149f9615fef8", size = 193786, upload-time = "2025-10-02T14:33:54.272Z" }, - { url = "https://files.pythonhosted.org/packages/c1/96/fec0be9bb4b8f5d9c57d76380a366f31a1781fb802f76fc7cda6c84893c7/xxhash-3.6.0-cp310-cp310-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:48e6f2ffb07a50b52465a1032c3cf1f4a5683f944acaca8a134a2f23674c2058", size = 212830, upload-time = "2025-10-02T14:33:55.706Z" }, - { url = "https://files.pythonhosted.org/packages/c4/a0/c706845ba77b9611f81fd2e93fad9859346b026e8445e76f8c6fd057cc6d/xxhash-3.6.0-cp310-cp310-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:b5b848ad6c16d308c3ac7ad4ba6bede80ed5df2ba8ed382f8932df63158dd4b2", size = 211606, upload-time = "2025-10-02T14:33:57.133Z" }, - { url = "https://files.pythonhosted.org/packages/67/1e/164126a2999e5045f04a69257eea946c0dc3e86541b400d4385d646b53d7/xxhash-3.6.0-cp310-cp310-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:a034590a727b44dd8ac5914236a7b8504144447a9682586c3327e935f33ec8cc", size = 444872, upload-time = "2025-10-02T14:33:58.446Z" }, - { url = "https://files.pythonhosted.org/packages/2d/4b/55ab404c56cd70a2cf5ecfe484838865d0fea5627365c6c8ca156bd09c8f/xxhash-3.6.0-cp310-cp310-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:8a8f1972e75ebdd161d7896743122834fe87378160c20e97f8b09166213bf8cc", size = 193217, upload-time = "2025-10-02T14:33:59.724Z" }, - { url = "https://files.pythonhosted.org/packages/45/e6/52abf06bac316db33aa269091ae7311bd53cfc6f4b120ae77bac1b348091/xxhash-3.6.0-cp310-cp310-musllinux_1_2_aarch64.whl", hash = "sha256:ee34327b187f002a596d7b167ebc59a1b729e963ce645964bbc050d2f1b73d07", size = 210139, upload-time = "2025-10-02T14:34:02.041Z" }, - { url = "https://files.pythonhosted.org/packages/34/37/db94d490b8691236d356bc249c08819cbcef9273a1a30acf1254ff9ce157/xxhash-3.6.0-cp310-cp310-musllinux_1_2_i686.whl", hash = "sha256:339f518c3c7a850dd033ab416ea25a692759dc7478a71131fe8869010d2b75e4", size = 197669, upload-time = "2025-10-02T14:34:03.664Z" }, - { url = "https://files.pythonhosted.org/packages/b7/36/c4f219ef4a17a4f7a64ed3569bc2b5a9c8311abdb22249ac96093625b1a4/xxhash-3.6.0-cp310-cp310-musllinux_1_2_ppc64le.whl", hash = "sha256:bf48889c9630542d4709192578aebbd836177c9f7a4a2778a7d6340107c65f06", size = 210018, upload-time = "2025-10-02T14:34:05.325Z" }, - { url = "https://files.pythonhosted.org/packages/fd/06/bfac889a374fc2fc439a69223d1750eed2e18a7db8514737ab630534fa08/xxhash-3.6.0-cp310-cp310-musllinux_1_2_s390x.whl", hash = "sha256:5576b002a56207f640636056b4160a378fe36a58db73ae5c27a7ec8db35f71d4", size = 413058, upload-time = "2025-10-02T14:34:06.925Z" }, - { url = "https://files.pythonhosted.org/packages/c9/d1/555d8447e0dd32ad0930a249a522bb2e289f0d08b6b16204cfa42c1f5a0c/xxhash-3.6.0-cp310-cp310-musllinux_1_2_x86_64.whl", hash = "sha256:af1f3278bd02814d6dedc5dec397993b549d6f16c19379721e5a1d31e132c49b", size = 190628, upload-time = "2025-10-02T14:34:08.669Z" }, - { url = "https://files.pythonhosted.org/packages/d1/15/8751330b5186cedc4ed4b597989882ea05e0408b53fa47bcb46a6125bfc6/xxhash-3.6.0-cp310-cp310-win32.whl", hash = "sha256:aed058764db109dc9052720da65fafe84873b05eb8b07e5e653597951af57c3b", size = 30577, upload-time = "2025-10-02T14:34:10.234Z" }, - { url = "https://files.pythonhosted.org/packages/bb/cc/53f87e8b5871a6eb2ff7e89c48c66093bda2be52315a8161ddc54ea550c4/xxhash-3.6.0-cp310-cp310-win_amd64.whl", hash = "sha256:e82da5670f2d0d98950317f82a0e4a0197150ff19a6df2ba40399c2a3b9ae5fb", size = 31487, upload-time = "2025-10-02T14:34:11.618Z" }, - { url = "https://files.pythonhosted.org/packages/9f/00/60f9ea3bb697667a14314d7269956f58bf56bb73864f8f8d52a3c2535e9a/xxhash-3.6.0-cp310-cp310-win_arm64.whl", hash = "sha256:4a082ffff8c6ac07707fb6b671caf7c6e020c75226c561830b73d862060f281d", size = 27863, upload-time = "2025-10-02T14:34:12.619Z" }, { url = "https://files.pythonhosted.org/packages/17/d4/cc2f0400e9154df4b9964249da78ebd72f318e35ccc425e9f403c392f22a/xxhash-3.6.0-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:b47bbd8cf2d72797f3c2772eaaac0ded3d3af26481a26d7d7d41dc2d3c46b04a", size = 32844, upload-time = "2025-10-02T14:34:14.037Z" }, { url = "https://files.pythonhosted.org/packages/5e/ec/1cc11cd13e26ea8bc3cb4af4eaadd8d46d5014aebb67be3f71fb0b68802a/xxhash-3.6.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:2b6821e94346f96db75abaa6e255706fb06ebd530899ed76d32cd99f20dc52fa", size = 30809, upload-time = "2025-10-02T14:34:15.484Z" }, { url = "https://files.pythonhosted.org/packages/04/5f/19fe357ea348d98ca22f456f75a30ac0916b51c753e1f8b2e0e6fb884cce/xxhash-3.6.0-cp311-cp311-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", hash = "sha256:d0a9751f71a1a65ce3584e9cae4467651c7e70c9d31017fa57574583a4540248", size = 194665, upload-time = "2025-10-02T14:34:16.541Z" }, @@ -2393,22 +2132,6 @@ version = "0.25.0" source = { registry = "https://pypi.org/simple" } sdist = { url = "https://files.pythonhosted.org/packages/fd/aa/3e0508d5a5dd96529cdc5a97011299056e14c6505b678fd58938792794b1/zstandard-0.25.0.tar.gz", hash = "sha256:7713e1179d162cf5c7906da876ec2ccb9c3a9dcbdffef0cc7f70c3667a205f0b", size = 711513, upload-time = "2025-09-14T22:15:54.002Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/56/7a/28efd1d371f1acd037ac64ed1c5e2b41514a6cc937dd6ab6a13ab9f0702f/zstandard-0.25.0-cp310-cp310-macosx_10_9_x86_64.whl", hash = "sha256:e59fdc271772f6686e01e1b3b74537259800f57e24280be3f29c8a0deb1904dd", size = 795256, upload-time = "2025-09-14T22:15:56.415Z" }, - { url = "https://files.pythonhosted.org/packages/96/34/ef34ef77f1ee38fc8e4f9775217a613b452916e633c4f1d98f31db52c4a5/zstandard-0.25.0-cp310-cp310-macosx_11_0_arm64.whl", hash = "sha256:4d441506e9b372386a5271c64125f72d5df6d2a8e8a2a45a0ae09b03cb781ef7", size = 640565, upload-time = "2025-09-14T22:15:58.177Z" }, - { url = "https://files.pythonhosted.org/packages/9d/1b/4fdb2c12eb58f31f28c4d28e8dc36611dd7205df8452e63f52fb6261d13e/zstandard-0.25.0-cp310-cp310-manylinux2010_i686.manylinux2014_i686.manylinux_2_12_i686.manylinux_2_17_i686.whl", hash = "sha256:ab85470ab54c2cb96e176f40342d9ed41e58ca5733be6a893b730e7af9c40550", size = 5345306, upload-time = "2025-09-14T22:16:00.165Z" }, - { url = "https://files.pythonhosted.org/packages/73/28/a44bdece01bca027b079f0e00be3b6bd89a4df180071da59a3dd7381665b/zstandard-0.25.0-cp310-cp310-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:e05ab82ea7753354bb054b92e2f288afb750e6b439ff6ca78af52939ebbc476d", size = 5055561, upload-time = "2025-09-14T22:16:02.22Z" }, - { url = "https://files.pythonhosted.org/packages/e9/74/68341185a4f32b274e0fc3410d5ad0750497e1acc20bd0f5b5f64ce17785/zstandard-0.25.0-cp310-cp310-manylinux2014_ppc64le.manylinux_2_17_ppc64le.whl", hash = "sha256:78228d8a6a1c177a96b94f7e2e8d012c55f9c760761980da16ae7546a15a8e9b", size = 5402214, upload-time = "2025-09-14T22:16:04.109Z" }, - { url = "https://files.pythonhosted.org/packages/8b/67/f92e64e748fd6aaffe01e2b75a083c0c4fd27abe1c8747fee4555fcee7dd/zstandard-0.25.0-cp310-cp310-manylinux2014_s390x.manylinux_2_17_s390x.whl", hash = "sha256:2b6bd67528ee8b5c5f10255735abc21aa106931f0dbaf297c7be0c886353c3d0", size = 5449703, upload-time = "2025-09-14T22:16:06.312Z" }, - { url = "https://files.pythonhosted.org/packages/fd/e5/6d36f92a197c3c17729a2125e29c169f460538a7d939a27eaaa6dcfcba8e/zstandard-0.25.0-cp310-cp310-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:4b6d83057e713ff235a12e73916b6d356e3084fd3d14ced499d84240f3eecee0", size = 5556583, upload-time = "2025-09-14T22:16:08.457Z" }, - { url = "https://files.pythonhosted.org/packages/d7/83/41939e60d8d7ebfe2b747be022d0806953799140a702b90ffe214d557638/zstandard-0.25.0-cp310-cp310-musllinux_1_1_aarch64.whl", hash = "sha256:9174f4ed06f790a6869b41cba05b43eeb9a35f8993c4422ab853b705e8112bbd", size = 5045332, upload-time = "2025-09-14T22:16:10.444Z" }, - { url = "https://files.pythonhosted.org/packages/b3/87/d3ee185e3d1aa0133399893697ae91f221fda79deb61adbe998a7235c43f/zstandard-0.25.0-cp310-cp310-musllinux_1_1_x86_64.whl", hash = "sha256:25f8f3cd45087d089aef5ba3848cd9efe3ad41163d3400862fb42f81a3a46701", size = 5572283, upload-time = "2025-09-14T22:16:12.128Z" }, - { url = "https://files.pythonhosted.org/packages/0a/1d/58635ae6104df96671076ac7d4ae7816838ce7debd94aecf83e30b7121b0/zstandard-0.25.0-cp310-cp310-musllinux_1_2_aarch64.whl", hash = "sha256:3756b3e9da9b83da1796f8809dd57cb024f838b9eeafde28f3cb472012797ac1", size = 4959754, upload-time = "2025-09-14T22:16:14.225Z" }, - { url = "https://files.pythonhosted.org/packages/75/d6/57e9cb0a9983e9a229dd8fd2e6e96593ef2aa82a3907188436f22b111ccd/zstandard-0.25.0-cp310-cp310-musllinux_1_2_i686.whl", hash = "sha256:81dad8d145d8fd981b2962b686b2241d3a1ea07733e76a2f15435dfb7fb60150", size = 5266477, upload-time = "2025-09-14T22:16:16.343Z" }, - { url = "https://files.pythonhosted.org/packages/d1/a9/ee891e5edf33a6ebce0a028726f0bbd8567effe20fe3d5808c42323e8542/zstandard-0.25.0-cp310-cp310-musllinux_1_2_ppc64le.whl", hash = "sha256:a5a419712cf88862a45a23def0ae063686db3d324cec7edbe40509d1a79a0aab", size = 5440914, upload-time = "2025-09-14T22:16:18.453Z" }, - { url = "https://files.pythonhosted.org/packages/58/08/a8522c28c08031a9521f27abc6f78dbdee7312a7463dd2cfc658b813323b/zstandard-0.25.0-cp310-cp310-musllinux_1_2_s390x.whl", hash = "sha256:e7360eae90809efd19b886e59a09dad07da4ca9ba096752e61a2e03c8aca188e", size = 5819847, upload-time = "2025-09-14T22:16:20.559Z" }, - { url = "https://files.pythonhosted.org/packages/6f/11/4c91411805c3f7b6f31c60e78ce347ca48f6f16d552fc659af6ec3b73202/zstandard-0.25.0-cp310-cp310-musllinux_1_2_x86_64.whl", hash = "sha256:75ffc32a569fb049499e63ce68c743155477610532da1eb38e7f24bf7cd29e74", size = 5363131, upload-time = "2025-09-14T22:16:22.206Z" }, - { url = "https://files.pythonhosted.org/packages/ef/d6/8c4bd38a3b24c4c7676a7a3d8de85d6ee7a983602a734b9f9cdefb04a5d6/zstandard-0.25.0-cp310-cp310-win32.whl", hash = "sha256:106281ae350e494f4ac8a80470e66d1fe27e497052c8d9c3b95dc4cf1ade81aa", size = 436469, upload-time = "2025-09-14T22:16:25.002Z" }, - { url = "https://files.pythonhosted.org/packages/93/90/96d50ad417a8ace5f841b3228e93d1bb13e6ad356737f42e2dde30d8bd68/zstandard-0.25.0-cp310-cp310-win_amd64.whl", hash = "sha256:ea9d54cc3d8064260114a0bbf3479fc4a98b21dffc89b3459edd506b69262f6e", size = 506100, upload-time = "2025-09-14T22:16:23.569Z" }, { url = "https://files.pythonhosted.org/packages/2a/83/c3ca27c363d104980f1c9cee1101cc8ba724ac8c28a033ede6aab89585b1/zstandard-0.25.0-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:933b65d7680ea337180733cf9e87293cc5500cc0eb3fc8769f4d3c88d724ec5c", size = 795254, upload-time = "2025-09-14T22:16:26.137Z" }, { url = "https://files.pythonhosted.org/packages/ac/4d/e66465c5411a7cf4866aeadc7d108081d8ceba9bc7abe6b14aa21c671ec3/zstandard-0.25.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:a3f79487c687b1fc69f19e487cd949bf3aae653d181dfb5fde3bf6d18894706f", size = 640559, upload-time = "2025-09-14T22:16:27.973Z" }, { url = "https://files.pythonhosted.org/packages/12/56/354fe655905f290d3b147b33fe946b0f27e791e4b50a5f004c802cb3eb7b/zstandard-0.25.0-cp311-cp311-manylinux2010_i686.manylinux2014_i686.manylinux_2_12_i686.manylinux_2_17_i686.whl", hash = "sha256:0bbc9a0c65ce0eea3c34a691e3c4b6889f5f3909ba4822ab385fab9057099431", size = 5348020, upload-time = "2025-09-14T22:16:29.523Z" }, From ee38a2bdce04897ed2c0c7c18e7335447ba368dc Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 6 Oct 2026 14:17:28 +0000 Subject: [PATCH 02/25] feat(openai)!: drop support for OpenAI SDK versions below 1.0 The OpenAI integration now always uses the v1 client shapes. Remove the pre-1.0 method definitions and every _is_openai_v1() branch, and raise the dev dependency floor to openai>=1.0.0. Co-authored-by: Hassieb Pakzad --- langfuse/openai.py | 63 +++++++++------------------------------------- pyproject.toml | 2 +- uv.lock | 2 +- 3 files changed, 14 insertions(+), 53 deletions(-) diff --git a/langfuse/openai.py b/langfuse/openai.py index 16e7f1c1f..d3758ecb5 100644 --- a/langfuse/openai.py +++ b/langfuse/openai.py @@ -80,24 +80,6 @@ class OpenAiDefinition: max_version: Optional[str] = None -OPENAI_METHODS_V0 = [ - OpenAiDefinition( - module="openai", - object="ChatCompletion", - method="create", - type="chat", - sync=True, - ), - OpenAiDefinition( - module="openai", - object="Completion", - method="create", - type="completion", - sync=True, - ), -] - - OPENAI_METHODS_V1 = [ OpenAiDefinition( module="openai.resources.chat.completions", @@ -798,8 +780,7 @@ def _extract_streamed_openai_response(resource: Any, chunks: Any) -> Any: model, usage, finish_reason, service_tier = None, None, None, None for chunk in chunks: - if _is_openai_v1(): - chunk = chunk.__dict__ + chunk = chunk.__dict__ model = model or chunk.get("model", None) or None service_tier = service_tier or chunk.get("service_tier", None) or None @@ -810,15 +791,14 @@ def _extract_streamed_openai_response(resource: Any, chunks: Any) -> Any: choices = chunk.get("choices") or [] for choice in choices: - if _is_openai_v1(): - choice = choice.__dict__ + choice = choice.__dict__ if resource.type == "chat": delta = choice.get("delta", None) choice_finish_reason = choice.get("finish_reason", None) if choice_finish_reason is not None: finish_reason = choice_finish_reason - if _is_openai_v1() and delta is not None: + if delta is not None: delta = delta.__dict__ if delta is None: @@ -956,7 +936,7 @@ def _get_langfuse_data_from_default_response( if len(choices) > 0: choice = choices[-1] - completion = choice.text if _is_openai_v1() else choice.get("text", None) + completion = choice.text elif resource.object == "Responses" or resource.object == "AsyncResponses": completion = _extract_response_api_completion(response.get("output", {})) @@ -968,17 +948,11 @@ def _get_langfuse_data_from_default_response( if len(choices) > 1: completion = [ _extract_chat_response(choice.message.__dict__) - if _is_openai_v1() - else choice.get("message", None) for choice in choices ] else: choice = choices[0] - completion = ( - _extract_chat_response(choice.message.__dict__) - if _is_openai_v1() - else choice.get("message", None) - ) + completion = _extract_chat_response(choice.message.__dict__) elif resource.type == "embedding": data = response.get("data") or [] @@ -1015,16 +989,12 @@ def _merge_service_tier_into_model_parameters( return {**(model_parameters or {}), "service_tier": service_tier} -def _is_openai_v1() -> bool: - return Version(openai.__version__) >= Version("1.0.0") - - def _is_streaming_response(response: Any) -> bool: return ( isinstance(response, types.GeneratorType) or isinstance(response, types.AsyncGeneratorType) - or (_is_openai_v1() and isinstance(response, openai.Stream)) - or (_is_openai_v1() and isinstance(response, openai.AsyncStream)) + or isinstance(response, openai.Stream) + or isinstance(response, openai.AsyncStream) ) @@ -1034,9 +1004,6 @@ def _is_streaming_response(response: Any) -> bool: def _install_openai_stream_iteration_hooks() -> None: global _openai_stream_iter_hook_installed - if not _is_openai_v1(): - return - if not _openai_stream_iter_hook_installed: original_iter = openai.Stream.__iter__ original_aiter = openai.AsyncStream.__aiter__ @@ -1318,7 +1285,7 @@ def _wrap( try: openai_response = wrapped(**arg_extractor.get_openai_args()) - if _is_openai_v1() and isinstance(openai_response, openai.Stream): + if isinstance(openai_response, openai.Stream): return _instrument_openai_stream( resource=open_ai_resource, response=openai_response, @@ -1338,9 +1305,7 @@ def _wrap( model, completion, usage, service_tier = ( _get_langfuse_data_from_default_response( open_ai_resource, - (parsed_response and parsed_response.__dict__) - if _is_openai_v1() - else parsed_response, + parsed_response and parsed_response.__dict__, ) ) @@ -1407,7 +1372,7 @@ async def _wrap_async( try: openai_response = await wrapped(**arg_extractor.get_openai_args()) - if _is_openai_v1() and isinstance(openai_response, openai.AsyncStream): + if isinstance(openai_response, openai.AsyncStream): return _instrument_openai_async_stream( resource=open_ai_resource, response=openai_response, @@ -1427,9 +1392,7 @@ async def _wrap_async( model, completion, usage, service_tier = ( _get_langfuse_data_from_default_response( open_ai_resource, - (parsed_response and parsed_response.__dict__) - if _is_openai_v1() - else parsed_response, + parsed_response and parsed_response.__dict__, ) ) generation.update( @@ -1460,9 +1423,7 @@ async def _wrap_async( def register_tracing() -> None: - resources = OPENAI_METHODS_V1 if _is_openai_v1() else OPENAI_METHODS_V0 - - for resource in resources: + for resource in OPENAI_METHODS_V1: if resource.min_version is not None and Version(openai.__version__) < Version( resource.min_version ): diff --git a/pyproject.toml b/pyproject.toml index 1534e4aa6..743bc32e6 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -52,7 +52,7 @@ dev = [ "pytest-httpserver>=1.0.8,<2", "ruff>=0.15.2,<0.16", "mypy>=1.0.0,<2", - "openai>=0.27.8", + "openai>=1.0.0", "langchain-openai>=0.0.5,<0.4", "langchain>=1,<2", "langgraph>=1,<2", diff --git a/uv.lock b/uv.lock index ecacc3a0d..701ea859a 100644 --- a/uv.lock +++ b/uv.lock @@ -560,7 +560,7 @@ dev = [ { name = "langchain-openai", specifier = ">=0.0.5,<0.4" }, { name = "langgraph", specifier = ">=1,<2" }, { name = "mypy", specifier = ">=1.0.0,<2" }, - { name = "openai", specifier = ">=0.27.8" }, + { name = "openai", specifier = ">=1.0.0" }, { name = "opentelemetry-instrumentation-threading", specifier = ">=0.59b0,<1" }, { name = "pre-commit", specifier = ">=3.2.2,<4" }, { name = "pytest", specifier = ">=7.4,<9.0" }, From 95b341dee7802220d67847217c61271fd8b66de3 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 6 Oct 2026 14:18:11 +0000 Subject: [PATCH 03/25] feat(otel)!: send x-langfuse-ingestion-version: 4 by default The default OTLP exporter now sends x-langfuse-ingestion-version: 4 so servers in dual-write or preview mode process exported spans directly. It is a no-op on events_only deployments, and additional_headers can still override it. Co-authored-by: Hassieb Pakzad --- langfuse/_client/span_processor.py | 1 + tests/unit/test_additional_headers_simple.py | 16 ++++++++++++++++ 2 files changed, 17 insertions(+) diff --git a/langfuse/_client/span_processor.py b/langfuse/_client/span_processor.py index b362b8da1..89b4f5bd6 100644 --- a/langfuse/_client/span_processor.py +++ b/langfuse/_client/span_processor.py @@ -159,6 +159,7 @@ def __init__( "x-langfuse-sdk-name": "python", "x-langfuse-sdk-version": langfuse_version, "x-langfuse-public-key": public_key, + "x-langfuse-ingestion-version": "4", } # Merge additional headers if provided diff --git a/tests/unit/test_additional_headers_simple.py b/tests/unit/test_additional_headers_simple.py index 7837c5c3e..fd7764fd6 100644 --- a/tests/unit/test_additional_headers_simple.py +++ b/tests/unit/test_additional_headers_simple.py @@ -210,6 +210,22 @@ def test_span_processor_none_additional_headers_works(self): assert "authorization" in exporter._client._headers assert "x-langfuse-sdk-name" in exporter._client._headers assert "x-langfuse-public-key" in exporter._client._headers + assert exporter._client._headers["x-langfuse-ingestion-version"] == "4" + + def test_span_processor_additional_headers_override_ingestion_version(self): + """Test that additional headers can override the default ingestion version.""" + from langfuse._client.span_processor import LangfuseSpanProcessor + + processor = LangfuseSpanProcessor( + public_key="test-public-key", + secret_key="test-secret-key", + base_url="https://mock-host.com", + additional_headers={"x-langfuse-ingestion-version": "3"}, + ) + + exporter = processor.span_exporter + + assert exporter._client._headers["x-langfuse-ingestion-version"] == "3" def test_span_processor_uses_custom_span_exporter_when_provided(self): """Test that a custom exporter bypasses the default OTLP exporter construction.""" From 466808b36bf073e5a2d87e6a3d0c7873cf23c642 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 6 Oct 2026 14:20:50 +0000 Subject: [PATCH 04/25] feat(client)!: remove blocked_instrumentation_scopes blocked_instrumentation_scopes was deprecated in favor of should_export_span. Remove it from Langfuse, the resource manager and the span processor, and drop the leftover is_active entry from the create_prompt docstring. LANGFUSE_HOST and host keep working, but now log a deprecation warning when they decide the base URL. Co-authored-by: Hassieb Pakzad --- langfuse/_client/client.py | 38 ++++---------- langfuse/_client/get_client.py | 1 - langfuse/_client/resource_manager.py | 5 -- langfuse/_client/span_processor.py | 27 +--------- tests/conftest.py | 4 -- tests/unit/test_app_root_detection.py | 7 ++- tests/unit/test_initialization.py | 30 +++++++++++ tests/unit/test_otel.py | 73 +-------------------------- 8 files changed, 46 insertions(+), 139 deletions(-) diff --git a/langfuse/_client/client.py b/langfuse/_client/client.py index af06e69ef..acf2a0c9e 100644 --- a/langfuse/_client/client.py +++ b/langfuse/_client/client.py @@ -9,7 +9,6 @@ import re import urllib.parse import uuid -import warnings from datetime import datetime from hashlib import sha256 from time import time_ns @@ -251,19 +250,6 @@ def mask_otel_spans( langfuse = Langfuse(mask_otel_spans=mask_otel_spans) ``` - blocked_instrumentation_scopes (Optional[List[str]]): Deprecated. Use `should_export_span` instead. Equivalent behavior: - ```python - from langfuse.span_filter import is_default_export_span - blocked = {"sqlite", "requests"} - - should_export_span = lambda span: ( - is_default_export_span(span) - and ( - span.instrumentation_scope is None - or span.instrumentation_scope.name not in blocked - ) - ) - ``` should_export_span (Optional[Callable[[ReadableSpan], bool]]): Callback to decide whether to export a span. If omitted, Langfuse uses the default filter (Langfuse SDK spans, spans with `gen_ai.*` attributes, and known LLM instrumentation scopes). additional_headers (Optional[Dict[str, str]]): Additional headers to include in all API requests and in the default OTLPSpanExporter requests. These headers will be merged with default headers. Note: If httpx_client is provided, additional_headers must be set directly on your custom httpx_client as well. If `span_exporter` is provided, these headers are not wired into that exporter and must be configured on the exporter instance directly. tracer_provider(Optional[TracerProvider]): OpenTelemetry TracerProvider to use for Langfuse. This can be useful to set to have disconnected tracing between Langfuse and other OpenTelemetry-span emitting libraries. Note: To track active spans, the context is still shared between TracerProviders. This may lead to broken trace trees. @@ -330,7 +316,6 @@ def __init__( sample_rate: Optional[float] = None, mask: Optional[MaskFunction] = None, mask_otel_spans: Optional[MaskOtelSpansFunction] = None, - blocked_instrumentation_scopes: Optional[List[str]] = None, should_export_span: Optional[Callable[[ReadableSpan], bool]] = None, additional_headers: Optional[Dict[str, str]] = None, tracer_provider: Optional[TracerProvider] = None, @@ -344,6 +329,15 @@ def __init__( or host or os.environ.get(LANGFUSE_HOST, "https://cloud.langfuse.com") ) + if ( + not base_url + and not os.environ.get(LANGFUSE_BASE_URL) + and (host or os.environ.get(LANGFUSE_HOST)) + ): + langfuse_logger.warning( + "`host` and LANGFUSE_HOST are deprecated. Use `base_url` or " + "LANGFUSE_BASE_URL instead." + ) self._environment = environment or cast( str, os.environ.get(LANGFUSE_TRACING_ENVIRONMENT) ) @@ -403,18 +397,6 @@ def __init__( "OTEL_SDK_DISABLED is set. Langfuse tracing will be disabled and no traces will appear in the UI." ) - if blocked_instrumentation_scopes is not None: - warnings.warn( - "`blocked_instrumentation_scopes` is deprecated and will be removed in a future release. " - "Use `should_export_span` instead. Example: " - "from langfuse.span_filter import is_default_export_span; " - 'blocked={"scope"}; should_export_span=lambda span: ' - "is_default_export_span(span) and (span.instrumentation_scope is None or " - "span.instrumentation_scope.name not in blocked).", - DeprecationWarning, - stacklevel=2, - ) - # Initialize api and tracer if requirements are met self._resources = LangfuseResourceManager( public_key=public_key, @@ -431,7 +413,6 @@ def __init__( mask=mask, mask_otel_spans=mask_otel_spans, tracing_enabled=self._tracing_enabled, - blocked_instrumentation_scopes=blocked_instrumentation_scopes, should_export_span=should_export_span, additional_headers=additional_headers, tracer_provider=tracer_provider, @@ -4137,7 +4118,6 @@ def create_prompt( Keyword Args: name : The name of the prompt to be created. prompt : The content of the prompt to be created. - is_active [DEPRECATED] : A flag indicating whether the prompt is active or not. This is deprecated and will be removed in a future release. Please use the 'production' label instead. labels: The labels of the prompt. Defaults to None. To create a default-served prompt, add the 'production' label. tags: The tags of the prompt. Defaults to None. Will be applied to all versions of the prompt. config: Additional structured data to be saved with the prompt. Defaults to None. diff --git a/langfuse/_client/get_client.py b/langfuse/_client/get_client.py index ff06c7d29..eceaed5ef 100644 --- a/langfuse/_client/get_client.py +++ b/langfuse/_client/get_client.py @@ -51,7 +51,6 @@ def _create_client_from_instance( sample_rate=instance.sample_rate, mask=instance.mask, mask_otel_spans=instance.mask_otel_spans, - blocked_instrumentation_scopes=instance.blocked_instrumentation_scopes, should_export_span=instance.should_export_span, additional_headers=instance.additional_headers, tracer_provider=instance.tracer_provider, diff --git a/langfuse/_client/resource_manager.py b/langfuse/_client/resource_manager.py index 4395872db..e1e091d31 100644 --- a/langfuse/_client/resource_manager.py +++ b/langfuse/_client/resource_manager.py @@ -127,7 +127,6 @@ def __new__( mask: Optional[MaskFunction] = None, mask_otel_spans: Optional[MaskOtelSpansFunction] = None, tracing_enabled: Optional[bool] = None, - blocked_instrumentation_scopes: Optional[List[str]] = None, should_export_span: Optional[Callable[[ReadableSpan], bool]] = None, additional_headers: Optional[Dict[str, str]] = None, tracer_provider: Optional[TracerProvider] = None, @@ -166,7 +165,6 @@ def __new__( tracing_enabled=tracing_enabled if tracing_enabled is not None else True, - blocked_instrumentation_scopes=blocked_instrumentation_scopes, should_export_span=should_export_span, additional_headers=additional_headers, tracer_provider=tracer_provider, @@ -196,7 +194,6 @@ def _initialize_instance( mask: Optional[MaskFunction] = None, mask_otel_spans: Optional[MaskOtelSpansFunction] = None, tracing_enabled: bool = True, - blocked_instrumentation_scopes: Optional[List[str]] = None, should_export_span: Optional[Callable[[ReadableSpan], bool]] = None, additional_headers: Optional[Dict[str, str]] = None, tracer_provider: Optional[TracerProvider] = None, @@ -220,7 +217,6 @@ def _initialize_instance( self.release = release self.media_upload_thread_count = media_upload_thread_count self.sample_rate = sample_rate - self.blocked_instrumentation_scopes = blocked_instrumentation_scopes self.should_export_span = should_export_span self.additional_headers = additional_headers self.id_generator = id_generator @@ -259,7 +255,6 @@ def _initialize_instance( timeout=timeout, flush_at=flush_at, flush_interval=flush_interval, - blocked_instrumentation_scopes=blocked_instrumentation_scopes, should_export_span=should_export_span, additional_headers=additional_headers, span_exporter=span_exporter, diff --git a/langfuse/_client/span_processor.py b/langfuse/_client/span_processor.py index 89b4f5bd6..ba2f5b2de 100644 --- a/langfuse/_client/span_processor.py +++ b/langfuse/_client/span_processor.py @@ -15,7 +15,7 @@ import logging import os import threading -from typing import Callable, Dict, List, Literal, Optional, cast +from typing import Callable, Dict, Literal, Optional, cast from opentelemetry import context as context_api from opentelemetry.context import Context @@ -121,7 +121,6 @@ def __init__( timeout: Optional[int] = None, flush_at: Optional[int] = None, flush_interval: Optional[float] = None, - blocked_instrumentation_scopes: Optional[List[str]] = None, should_export_span: Optional[Callable[[ReadableSpan], bool]] = None, additional_headers: Optional[Dict[str, str]] = None, span_exporter: Optional[SpanExporter] = None, @@ -130,11 +129,6 @@ def __init__( otel_compression: Optional[Literal["gzip", "none"]] = None, ): self.public_key = public_key - self.blocked_instrumentation_scopes = ( - blocked_instrumentation_scopes - if blocked_instrumentation_scopes is not None - else [] - ) self._should_export_span = should_export_span or is_default_export_span self._app_root_lock = threading.Lock() @@ -258,16 +252,6 @@ def on_end(self, span: ReadableSpan) -> None: ) return - # Do not export spans from blocked instrumentation scopes - if self._is_blocked_instrumentation_scope(span): - langfuse_logger.debug( - "Trace: Dropping span due to blocked instrumentation scope | " - "span_name='%s' | instrumentation_scope='%s'", - span.name, - self._get_scope_name(span), - ) - return - # Apply custom or default span filter try: should_export = self._should_export_span(span) @@ -342,9 +326,6 @@ def _is_expected_exported_at_start(self, span: Span) -> bool: ): return False - if self._is_blocked_instrumentation_scope(readable_span): - return False - try: return bool(self._should_export_span(readable_span)) except Exception as error: @@ -359,12 +340,6 @@ def _is_expected_exported_at_start(self, span: Span) -> bool: return False - def _is_blocked_instrumentation_scope(self, span: ReadableSpan) -> bool: - return ( - span.instrumentation_scope is not None - and span.instrumentation_scope.name in self.blocked_instrumentation_scopes - ) - def _is_langfuse_project_span(self, span: ReadableSpan) -> bool: if not is_langfuse_span(span): return False diff --git a/tests/conftest.py b/tests/conftest.py index 0163842c6..97a78d212 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -102,10 +102,6 @@ def mock_init(self: Any, **kwargs: Any) -> None: from langfuse._client.span_filter import is_default_export_span self.public_key = kwargs.get("public_key", "test-public-key") - blocked_scopes = kwargs.get("blocked_instrumentation_scopes") - self.blocked_instrumentation_scopes = ( - blocked_scopes if blocked_scopes is not None else [] - ) self._should_export_span = ( kwargs.get("should_export_span") or is_default_export_span ) diff --git a/tests/unit/test_app_root_detection.py b/tests/unit/test_app_root_detection.py index 7fa346d1e..fdf19639a 100644 --- a/tests/unit/test_app_root_detection.py +++ b/tests/unit/test_app_root_detection.py @@ -252,12 +252,15 @@ def test_active_langfuse_scope_sets_baggage_after_root_start( assert "langfuse.trace.metadata.trace_id" not in spans["child"].attributes -def test_blocked_instrumentation_scope_parent_marks_child_as_app_root( +def test_filtered_scope_parent_marks_child_as_app_root( memory_exporter, ): tracer_provider, processor = _create_processor( memory_exporter, - blocked_instrumentation_scopes=["blocked.scope"], + should_export_span=lambda span: ( + span.instrumentation_scope is None + or span.instrumentation_scope.name != "blocked.scope" + ), ) blocked_tracer = tracer_provider.get_tracer("blocked.scope") langfuse_tracer = _langfuse_tracer(tracer_provider) diff --git a/tests/unit/test_initialization.py b/tests/unit/test_initialization.py index 7181ae45e..f564aee47 100644 --- a/tests/unit/test_initialization.py +++ b/tests/unit/test_initialization.py @@ -100,6 +100,36 @@ def test_env_host_fallback(self, cleanup_env_vars): assert client._base_url == "http://env-host.com" + def test_env_host_fallback_logs_deprecation_warning(self, cleanup_env_vars, caplog): + """Test that resolving the URL from LANGFUSE_HOST logs a deprecation warning.""" + caplog.set_level("WARNING", logger="langfuse") + os.environ["LANGFUSE_HOST"] = "http://env-host.com" + + Langfuse(public_key="test_pk", secret_key="test_sk") + + assert any( + "LANGFUSE_HOST are deprecated" in record.message + for record in caplog.records + ) + + def test_host_ignored_when_base_url_set_logs_no_warning( + self, cleanup_env_vars, caplog + ): + """Test that a set base URL suppresses the host deprecation warning.""" + caplog.set_level("WARNING", logger="langfuse") + os.environ["LANGFUSE_HOST"] = "http://env-host.com" + + Langfuse( + base_url="http://param-base-url.com", + public_key="test_pk", + secret_key="test_sk", + ) + + assert not any( + "LANGFUSE_HOST are deprecated" in record.message + for record in caplog.records + ) + def test_default_base_url(self, cleanup_env_vars): """Test that default base_url is used when nothing is set.""" client = Langfuse( diff --git a/tests/unit/test_otel.py b/tests/unit/test_otel.py index 46a085a71..ccdec39fd 100644 --- a/tests/unit/test_otel.py +++ b/tests/unit/test_otel.py @@ -120,10 +120,6 @@ def mock_init(self, **kwargs): from langfuse._client.span_filter import is_default_export_span self.public_key = kwargs.get("public_key", "test-key") - blocked_scopes = kwargs.get("blocked_instrumentation_scopes") - self.blocked_instrumentation_scopes = ( - blocked_scopes if blocked_scopes is not None else [] - ) self._should_export_span = ( kwargs.get("should_export_span") or is_default_export_span ) @@ -2408,10 +2404,6 @@ def mock_processor_init(self, **kwargs): from langfuse._client.span_filter import is_default_export_span self.public_key = kwargs.get("public_key", "test-key") - blocked_scopes = kwargs.get("blocked_instrumentation_scopes") - self.blocked_instrumentation_scopes = ( - blocked_scopes if blocked_scopes is not None else [] - ) self._should_export_span = ( kwargs.get("should_export_span") or is_default_export_span ) @@ -2446,16 +2438,13 @@ def mock_initialize(self, **kwargs): # Call original_initialize to set up all the necessary attributes original_initialize(self, **kwargs) - # Now create our custom LangfuseSpanProcessor with the actual blocked_instrumentation_scopes + # Now create our custom LangfuseSpanProcessor with the actual should_export_span from langfuse._client.span_processor import LangfuseSpanProcessor processor = LangfuseSpanProcessor( public_key=self.public_key, secret_key=self.secret_key, base_url=self.base_url, - blocked_instrumentation_scopes=kwargs.get( - "blocked_instrumentation_scopes" - ), should_export_span=kwargs.get("should_export_span"), ) # Replace its exporter with our test exporter @@ -2661,38 +2650,6 @@ def test_custom_should_export_span_with_composition( assert "known-span" in exported_span_names assert "unknown-span" not in exported_span_names - def test_blocked_scopes_override_should_export( - self, instrumentation_filtering_setup - ): - """Test that blocked scopes are dropped even when callback allows all.""" - with pytest.warns(DeprecationWarning, match="blocked_instrumentation_scopes"): - Langfuse( - public_key=instrumentation_filtering_setup["test_key"], - secret_key="test-secret-key", - base_url="http://localhost:3000", - blocked_instrumentation_scopes=["my-framework.worker"], - should_export_span=lambda span: True, - ) - - tracer_provider = instrumentation_filtering_setup["test_tracer_provider"] - blocked_tracer = tracer_provider.get_tracer("my-framework.worker") - allowed_tracer = tracer_provider.get_tracer("custom.allowed") - - blocked_span = blocked_tracer.start_span("blocked-span") - blocked_span.end() - allowed_span = allowed_tracer.start_span("allowed-span") - allowed_span.end() - tracer_provider.force_flush() - - exported_span_names = [ - span.name - for span in instrumentation_filtering_setup[ - "blocked_exporter" - ].get_finished_spans() - ] - assert "blocked-span" not in exported_span_names - assert "allowed-span" in exported_span_names - def test_should_export_span_with_none_uses_default( self, instrumentation_filtering_setup ): @@ -2758,34 +2715,6 @@ def _failing_filter(_span): for record in caplog.records ) - def test_blocked_scope_drop_logs_scope_name( - self, instrumentation_filtering_setup, caplog - ): - """Test that blocked scope drops include scope names in debug logs.""" - caplog.set_level("DEBUG", logger="langfuse") - - with pytest.warns(DeprecationWarning, match="blocked_instrumentation_scopes"): - Langfuse( - public_key=instrumentation_filtering_setup["test_key"], - secret_key="test-secret-key", - base_url="http://localhost:3000", - blocked_instrumentation_scopes=["my.blocked.scope"], - should_export_span=lambda span: True, - ) - - tracer_provider = instrumentation_filtering_setup["test_tracer_provider"] - blocked_tracer = tracer_provider.get_tracer("my.blocked.scope") - - span = blocked_tracer.start_span("blocked-debug-span") - span.end() - tracer_provider.force_flush() - - assert any( - "Dropping span due to blocked instrumentation scope" in record.message - and "my.blocked.scope" in record.message - for record in caplog.records - ) - class TestConcurrencyAndAsync(TestOTelBase): """Tests for asynchronous and concurrent span operations.""" From e9623ee341fbc19371acbb5c998f54e20b90e454 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 6 Oct 2026 14:22:16 +0000 Subject: [PATCH 05/25] feat(tracing)!: JSON-serialize propagated metadata values propagate_attributes(metadata=...) coerced non-string values with str(), so a boolean became "True" and a list became its Python repr. Serialize them as JSON instead, like observation metadata, and drop None values instead of sending "None". Co-authored-by: Hassieb Pakzad --- langfuse/_client/propagation.py | 18 ++++++++----- tests/unit/test_propagate_attributes.py | 35 ++++++++++++++++++++++--- 2 files changed, 43 insertions(+), 10 deletions(-) diff --git a/langfuse/_client/propagation.py b/langfuse/_client/propagation.py index ecf961e4f..1746554b8 100644 --- a/langfuse/_client/propagation.py +++ b/langfuse/_client/propagation.py @@ -38,7 +38,7 @@ _agnosticcontextmanager, ) -from langfuse._client.attributes import LangfuseOtelSpanAttributes +from langfuse._client.attributes import LangfuseOtelSpanAttributes, _serialize from langfuse._client.constants import LANGFUSE_SDK_EXPERIMENT_ENVIRONMENT from langfuse.logger import langfuse_logger from langfuse.model import PromptClient @@ -281,8 +281,9 @@ def propagate_attributes( - **Validation**: Attribute values (user_id, session_id, version, tags, trace_name) must be strings ≤200 characters. Environment must also match Langfuse's environment format: lowercase alphanumeric with optional - hyphens or underscores, must be ≤40 characters, and it must not start with "langfuse". Metadata - values are coerced to strings before the 200 character limit is applied. + hyphens or underscores, must be ≤40 characters, and it must not start with "langfuse". Non-string + metadata values are JSON-serialized before the 200 character limit is + applied, and None values are dropped. Invalid values will be dropped with a warning logged. - **OpenTelemetry**: This uses OpenTelemetry context propagation under the hood, making it compatible with other OTel-instrumented libraries. @@ -393,10 +394,15 @@ def _propagate_attributes( validated_metadata: Dict[str, str] = {} for key, value in metadata_value.items(): - coerced_value = value if isinstance(value, str) else str(value) + serialized_value = _serialize(value) - if _validate_string_value(value=coerced_value, key=f"{metadata_key}.{key}"): - validated_metadata[key] = coerced_value + if serialized_value is None: + continue + + if _validate_string_value( + value=serialized_value, key=f"{metadata_key}.{key}" + ): + validated_metadata[key] = serialized_value if validated_metadata: context = _set_propagated_attribute( diff --git a/tests/unit/test_propagate_attributes.py b/tests/unit/test_propagate_attributes.py index cdf4f9351..c49a4e956 100644 --- a/tests/unit/test_propagate_attributes.py +++ b/tests/unit/test_propagate_attributes.py @@ -461,10 +461,10 @@ def test_non_string_user_id_dropped(self, langfuse_client, memory_exporter): child_span, LangfuseOtelSpanAttributes.TRACE_USER_ID ) - def test_non_string_metadata_values_coerced( + def test_non_string_metadata_values_json_serialized( self, langfuse_client, memory_exporter, caplog ): - """Verify non-string metadata values are coerced instead of dropped.""" + """Verify non-string metadata values are JSON-serialized instead of dropped.""" caplog.set_level("WARNING", logger="langfuse") metadata = { @@ -472,6 +472,18 @@ def test_non_string_metadata_values_coerced( "langgraph_triggers": ["branch:agent"], "langgraph_path": ("root", "agent"), "max_search_results": 5, + "is_cached": True, + "ratio": 0.5, + "config": {"model": "gpt-4o"}, + } + expected = { + "langgraph_step": "1", + "langgraph_triggers": '["branch:agent"]', + "langgraph_path": '["root", "agent"]', + "max_search_results": "5", + "is_cached": "true", + "ratio": "0.5", + "config": '{"model": "gpt-4o"}', } with langfuse_client.start_as_current_observation(name="parent-span"): @@ -481,15 +493,30 @@ def test_non_string_metadata_values_coerced( child_span = self.get_span_by_name(memory_exporter, "child-span") - for key, value in metadata.items(): + for key, value in expected.items(): self.verify_span_attribute( child_span, f"{LangfuseOtelSpanAttributes.TRACE_METADATA}.{key}", - str(value), + value, ) assert "value is not a string. Dropping value." not in caplog.text + def test_none_metadata_values_dropped(self, langfuse_client, memory_exporter): + """Verify None metadata values are dropped instead of sent as 'None'.""" + with langfuse_client.start_as_current_observation(name="parent-span"): + with propagate_attributes(metadata={"kept": "yes", "empty": None}): + child = langfuse_client.start_observation(name="child-span") + child.end() + + child_span = self.get_span_by_name(memory_exporter, "child-span") + self.verify_span_attribute( + child_span, f"{LangfuseOtelSpanAttributes.TRACE_METADATA}.kept", "yes" + ) + self.verify_missing_attribute( + child_span, f"{LangfuseOtelSpanAttributes.TRACE_METADATA}.empty" + ) + def test_mixed_valid_invalid_metadata(self, langfuse_client, memory_exporter): """Verify mixed valid/invalid metadata - valid entries kept, invalid dropped.""" with langfuse_client.start_as_current_observation(name="parent-span"): From 66f9842b884f65724f50adbced03c15457b46f24 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 6 Oct 2026 14:22:56 +0000 Subject: [PATCH 06/25] feat(client)!: remove get_dataset_run, get_dataset_runs and delete_dataset_run These helpers call the dataset-run endpoints that Langfuse v4 no longer serves, so they always fail with a 404. Read experiment runs through langfuse.api.experiments.list() and list_items() instead. Co-authored-by: Hassieb Pakzad --- langfuse/_client/client.py | 84 -------------------------------------- 1 file changed, 84 deletions(-) diff --git a/langfuse/_client/client.py b/langfuse/_client/client.py index acf2a0c9e..8f7df6623 100644 --- a/langfuse/_client/client.py +++ b/langfuse/_client/client.py @@ -102,14 +102,11 @@ Dataset, DatasetItem, DatasetItemMediaReferenceField, - DatasetRunWithItems, DatasetStatus, - DeleteDatasetRunResponse, Error, LangfuseAPI, MapValue, NotFoundError, - PaginatedDatasetRuns, Prompt_Chat, Prompt_Text, ScoreBody, @@ -2513,87 +2510,6 @@ def get_dataset( handle_fern_exception(e) raise e - def get_dataset_run( - self, *, dataset_name: str, run_name: str - ) -> DatasetRunWithItems: - """Fetch a dataset run by dataset name and run name. - - Args: - dataset_name (str): The name of the dataset. - run_name (str): The name of the run. - - Returns: - DatasetRunWithItems: The dataset run with its items. - """ - try: - return cast( - DatasetRunWithItems, - self.api.datasets.get_run( - dataset_name=self._url_encode(dataset_name), - run_name=self._url_encode(run_name), - request_options=None, - ), - ) - except Error as e: - handle_fern_exception(e) - raise e - - def get_dataset_runs( - self, - *, - dataset_name: str, - page: Optional[int] = None, - limit: Optional[int] = None, - ) -> PaginatedDatasetRuns: - """Fetch all runs for a dataset. - - Args: - dataset_name (str): The name of the dataset. - page (Optional[int]): Page number, starts at 1. - limit (Optional[int]): Limit of items per page. - - Returns: - PaginatedDatasetRuns: Paginated list of dataset runs. - """ - try: - return cast( - PaginatedDatasetRuns, - self.api.datasets.get_runs( - dataset_name=self._url_encode(dataset_name), - page=page, - limit=limit, - request_options=None, - ), - ) - except Error as e: - handle_fern_exception(e) - raise e - - def delete_dataset_run( - self, *, dataset_name: str, run_name: str - ) -> DeleteDatasetRunResponse: - """Delete a dataset run and all its run items. This action is irreversible. - - Args: - dataset_name (str): The name of the dataset. - run_name (str): The name of the run. - - Returns: - DeleteDatasetRunResponse: Confirmation of deletion. - """ - try: - return cast( - DeleteDatasetRunResponse, - self.api.datasets.delete_run( - dataset_name=self._url_encode(dataset_name), - run_name=self._url_encode(run_name), - request_options=None, - ), - ) - except Error as e: - handle_fern_exception(e) - raise e - def run_experiment( self, *, From 6c5c4884ecf2874488ad0d39b1e717cc99091249 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 6 Oct 2026 14:20:37 +0000 Subject: [PATCH 07/25] feat(experiments)!: record experiment runs via OTEL only Generate the experiment id client-side instead of creating dataset run items via POST /dataset-run-items. Dataset-backed runs derive a stable id from project, dataset and run name (matching the server's derivation), so re-runs with the same run name keep grouping into one experiment. Local runs keep a random id. - set langfuse.experiment.item.version from the pinned dataset version - always persist run evaluations as scores on the experiment, incl. local data - return experiment_id / experiment_url (v4 results page) on results; dataset_run_id / dataset_run_url remain as deprecated aliases BREAKING CHANGE: run_experiment no longer creates Postgres dataset runs; dataset_run_id / dataset_run_url are deprecated aliases that are now also set for local-data experiments, and the URL points to the experiments page. Co-authored-by: Hassieb Pakzad --- langfuse/_client/attributes.py | 1 + langfuse/_client/client.py | 190 ++++++++++++--------- langfuse/_client/datasets.py | 41 ++--- langfuse/experiment.py | 77 ++++++--- tests/unit/test_experiment.py | 215 +++++++++++++++++++++++- tests/unit/test_propagate_attributes.py | 38 ++--- 6 files changed, 416 insertions(+), 146 deletions(-) diff --git a/langfuse/_client/attributes.py b/langfuse/_client/attributes.py index 4660b50f0..43a85c2fd 100644 --- a/langfuse/_client/attributes.py +++ b/langfuse/_client/attributes.py @@ -68,6 +68,7 @@ class LangfuseOtelSpanAttributes: EXPERIMENT_METADATA = "langfuse.experiment.metadata" EXPERIMENT_DATASET_ID = "langfuse.experiment.dataset.id" EXPERIMENT_ITEM_ID = "langfuse.experiment.item.id" + EXPERIMENT_ITEM_VERSION = "langfuse.experiment.item.version" EXPERIMENT_ITEM_EXPECTED_OUTPUT = "langfuse.experiment.item.expected_output" EXPERIMENT_ITEM_METADATA = "langfuse.experiment.item.metadata" EXPERIMENT_ITEM_ROOT_OBSERVATION_ID = "langfuse.experiment.item.root_observation_id" diff --git a/langfuse/_client/client.py b/langfuse/_client/client.py index 8f7df6623..653243b7d 100644 --- a/langfuse/_client/client.py +++ b/langfuse/_client/client.py @@ -2542,8 +2542,9 @@ def run_experiment( Args: name: Human-readable name for the experiment. Used for identification in the Langfuse UI. - run_name: Optional exact name for the experiment run. If provided, this will be - used as the exact dataset run name if the `data` contains Langfuse dataset items. + run_name: Optional exact name for the experiment run. If the `data` contains + Langfuse dataset items, runs with the same `run_name` on the same dataset + share one experiment ID and appear as a single experiment in Langfuse. If not provided, this will default to the experiment name appended with an ISO timestamp. description: Optional description explaining the experiment's purpose, methodology, or expected outcomes. @@ -2567,17 +2568,22 @@ def run_experiment( Controls the number of items processed simultaneously. Adjust based on API rate limits and system resources. metadata: Optional metadata dictionary to attach to all experiment traces. - This metadata will be included in every trace created during the experiment. - If `data` are Langfuse dataset items, the metadata will be attached to the dataset run, too. + This metadata will be included in every trace created during the experiment + and is shown as the experiment's metadata in Langfuse. Returns: ExperimentResult containing: - - run_name: The experiment run name. This is equal to the dataset run name if experiment was on Langfuse dataset. + - run_name: The experiment run name, shown as the experiment name in Langfuse. - item_results: List of results for each processed item with outputs and evaluations - - run_evaluations: List of aggregate evaluation results for the entire run - - experiment_id: Stable identifier for the experiment run across all items - - dataset_run_id: ID of the dataset run (if using Langfuse datasets) - - dataset_run_url: Direct URL to view results in Langfuse UI (if applicable) + - run_evaluations: List of aggregate evaluation results for the entire run. + They are stored as scores on the experiment run, also for local data. + - experiment_id: Identifier for the experiment run across all items. For + Langfuse dataset items it is derived from the project ID, dataset ID and + run name; for local data it is random. + - experiment_url: Direct URL to view results in Langfuse UI (None if the + project ID could not be resolved) + - dataset_run_id / dataset_run_url: Deprecated aliases of experiment_id / + experiment_url Raises: ValueError: If required parameters are missing or invalid @@ -2675,14 +2681,16 @@ def average_accuracy(*, item_results, **kwargs): ) # Results automatically linked to dataset in Langfuse UI - print(f"View results: {result['dataset_run_url']}") + print(f"View results: {result.experiment_url}") ``` Note: - Task and evaluator functions can be either synchronous or asynchronous - Individual item failures are logged but don't stop the experiment - All executions are automatically traced and visible in Langfuse UI - - When using Langfuse datasets, results are automatically linked for easy comparison + - Experiment runs are recorded only through the exported traces; no + separate dataset run is created via the API + - When using Langfuse datasets, results are automatically linked to the dataset for easy comparison - This method works in both sync and async contexts (Jupyter notebooks, web apps, etc.) - Async execution is handled automatically with smart event loop detection """ @@ -2726,7 +2734,33 @@ async def _run_experiment_async( "Starting experiment '%s' run '%s' with %s items", name, run_name, len(data) ) - shared_fallback_experiment_id = self._create_observation_id() + try: + project_id = await asyncio.to_thread(self._get_project_id) + except Exception as e: + langfuse_logger.warning( + "Failed to resolve project id for experiment: %s", e + ) + project_id = None + + # One experiment id per run: mixed-dataset data uses the first dataset item's dataset. + experiment_dataset_id = next( + ( + getattr(item, "dataset_id", None) + for item in data + if not isinstance(item, dict) and getattr(item, "dataset_id", None) + ), + None, + ) + experiment_id = self._create_experiment_id( + project_id=project_id, + dataset_id=experiment_dataset_id, + run_name=run_name, + ) + experiment_item_version = ( + self._format_experiment_item_version(dataset_version) + if dataset_version is not None + else None + ) # Set up concurrency control semaphore = asyncio.Semaphore(max_concurrency) @@ -2739,12 +2773,12 @@ async def process_item(item: ExperimentItem) -> ExperimentItemResult: task, evaluators, composite_evaluator, - shared_fallback_experiment_id, + experiment_id, name, run_name, description, metadata, - dataset_version, + experiment_item_version, ) # Run all items concurrently @@ -2770,47 +2804,24 @@ async def process_item(item: ExperimentItem) -> ExperimentItemResult: except Exception as e: langfuse_logger.error("Run evaluator failed: %s", e) - # Generate dataset run URL if applicable - dataset_run_id = next( - ( - result.dataset_run_id - for result in valid_results - if result.dataset_run_id - ), - None, + experiment_url = ( + f"{self._base_url}/project/{project_id}/experiments/results?baseline={experiment_id}" + if project_id + else None ) - dataset_run_url = None - if dataset_run_id and data: - try: - # Check if the first item has dataset_id (for DatasetItem objects) - first_item = data[0] - dataset_id = None - if hasattr(first_item, "dataset_id"): - dataset_id = getattr(first_item, "dataset_id", None) - - if dataset_id: - project_id = self._get_project_id() - - if project_id: - dataset_run_url = f"{self._base_url}/project/{project_id}/datasets/{dataset_id}/runs/{dataset_run_id}" - - except Exception: - pass # URL generation is optional - - # Store run-level evaluations as scores + # Run-level scores attach to the experiment via dataset_run_id == experiment_id. for evaluation in run_evaluations: try: - if dataset_run_id: - self.create_score( - dataset_run_id=dataset_run_id, - name=evaluation.name or "", - value=evaluation.value, # type: ignore - comment=evaluation.comment, - metadata=evaluation.metadata, - data_type=evaluation.data_type, # type: ignore - config_id=evaluation.config_id, - ) + self.create_score( + dataset_run_id=experiment_id, + name=evaluation.name or "", + value=evaluation.value, # type: ignore + comment=evaluation.comment, + metadata=evaluation.metadata, + data_type=evaluation.data_type, # type: ignore + config_id=evaluation.config_id, + ) except Exception as e: langfuse_logger.error("Failed to store run evaluation: %s", e) @@ -2824,23 +2835,58 @@ async def process_item(item: ExperimentItem) -> ExperimentItemResult: description=description, item_results=valid_results, run_evaluations=run_evaluations, - experiment_id=dataset_run_id or shared_fallback_experiment_id, - dataset_run_id=dataset_run_id, - dataset_run_url=dataset_run_url, + experiment_id=experiment_id, + experiment_url=experiment_url, ) + def _create_experiment_id( + self, + *, + project_id: Optional[str], + dataset_id: Optional[str], + run_name: str, + ) -> str: + if dataset_id is None: + return self._create_observation_id() + + if project_id is None: + langfuse_logger.warning( + "Could not resolve the project id; using a random experiment id for run '%s'. " + "Re-runs with the same run name will not be grouped into the same experiment.", + run_name, + ) + return self._create_observation_id() + + import json + + # Must match the Langfuse server's stable experiment id derivation byte for + # byte (JS JSON.stringify of the same array, SHA-256, first 16 hex chars). + payload = json.dumps( + ["langfuse-experiment-v1", project_id, dataset_id, run_name], + separators=(",", ":"), + ensure_ascii=False, + ) + + return sha256(payload.encode("utf-8")).hexdigest()[:16] + + @staticmethod + def _format_experiment_item_version(dataset_version: datetime) -> str: + from langfuse.api.core.datetime_utils import serialize_datetime + + return serialize_datetime(dataset_version) + async def _process_experiment_item( self, item: ExperimentItem, task: Callable, evaluators: List[Callable], composite_evaluator: Optional[CompositeEvaluatorFunction], - fallback_experiment_id: str, + experiment_id: str, experiment_name: str, experiment_run_name: str, experiment_description: Optional[str], experiment_metadata: Optional[Dict[str, Any]] = None, - dataset_version: Optional[datetime] = None, + experiment_item_version: Optional[str] = None, ) -> ExperimentItemResult: with self.start_as_current_observation(name="experiment-item-run") as span: try: @@ -2875,7 +2921,6 @@ async def _process_experiment_item( trace_id = span.trace_id dataset_id = None dataset_item_id = None - dataset_run_id = None if ( not isinstance(item, dict) @@ -2900,6 +2945,9 @@ async def _process_experiment_item( LangfuseOtelSpanAttributes.EXPERIMENT_ITEM_EXPECTED_OUTPUT: _serialize( expected_output ), + LangfuseOtelSpanAttributes.EXPERIMENT_ITEM_VERSION: ( + experiment_item_version if dataset_id else None + ), }.items() if v is not None } @@ -2913,32 +2961,6 @@ async def _process_experiment_item( ) as task_span: task_span._otel_span.set_attributes(experiment_span_attributes) - # Link dataset runs to the canonical task observation so their - # latency excludes the subsequent evaluator subtree. - if hasattr(item, "id") and hasattr(item, "dataset_id"): - try: - # Use sync API to avoid event loop issues when - # run_async_safely creates multiple event loops across - # different threads. - dataset_run_item = await asyncio.to_thread( - self.api.dataset_run_items.create, - run_name=experiment_run_name, - run_description=experiment_description, - metadata=experiment_metadata, - dataset_item_id=item.id, # type: ignore - trace_id=trace_id, - observation_id=task_span.id, - dataset_version=dataset_version, - ) - - dataset_run_id = dataset_run_item.dataset_run_id - - except Exception as e: - langfuse_logger.error( - "Failed to create dataset run item: %s", e - ) - - experiment_id = dataset_run_id or fallback_experiment_id propagated_experiment_attributes = PropagatedExperimentAttributes( experiment_id=experiment_id, experiment_name=experiment_run_name, @@ -3140,7 +3162,7 @@ async def _process_experiment_item( output=output, evaluations=evaluations, trace_id=trace_id, - dataset_run_id=dataset_run_id, + experiment_id=experiment_id, ) def _create_experiment_run_name( diff --git a/langfuse/_client/datasets.py b/langfuse/_client/datasets.py index e28ddbebf..1070c76d3 100644 --- a/langfuse/_client/datasets.py +++ b/langfuse/_client/datasets.py @@ -94,11 +94,11 @@ def run_experiment( """Run an experiment on this Langfuse dataset with automatic tracking. This is a convenience method that runs an experiment using all items in this - dataset. It automatically creates a dataset run in Langfuse for tracking and - comparison purposes, linking all experiment results to the dataset. + dataset. All results are recorded as one experiment in Langfuse that is linked + to the dataset for tracking and comparison purposes. Key benefits of using dataset.run_experiment(): - - Automatic dataset run creation and linking in Langfuse UI + - Automatic experiment tracking and linking to the dataset in Langfuse UI - Built-in experiment tracking and versioning - Easy comparison between different experiment runs - Direct access to dataset items with their metadata and expected outputs @@ -106,10 +106,11 @@ def run_experiment( Args: name: Human-readable name for the experiment run. This will be used as - the dataset run name in Langfuse for tracking and identification. - run_name: Optional exact name for the dataset run. If provided, this will be - used as the exact dataset run name in Langfuse. If not provided, this will - default to the experiment name appended with an ISO timestamp. + the base of the experiment run name in Langfuse for tracking and identification. + run_name: Optional exact name for the experiment run. Runs with the same + `run_name` on this dataset share one experiment ID and appear as a single + experiment in Langfuse. If not provided, this will default to the + experiment name appended with an ISO timestamp. description: Optional description of the experiment's purpose, methodology, or what you're testing. Appears in the Langfuse UI for context. task: Function that processes each dataset item and returns output. @@ -131,12 +132,14 @@ def run_experiment( Returns: ExperimentResult object containing: - name: The experiment name. - - run_name: The experiment run name (equivalent to the dataset run name). + - run_name: The experiment run name, shown as the experiment name in Langfuse. - description: Optional experiment description. - item_results: Results for each dataset item with outputs and evaluations. - run_evaluations: Aggregate evaluation results for the entire run. - - dataset_run_id: ID of the created dataset run in Langfuse. - - dataset_run_url: Direct URL to view the experiment results in Langfuse UI. + - experiment_id: ID of the experiment run in Langfuse. + - experiment_url: Direct URL to view the experiment results in Langfuse UI. + - dataset_run_id / dataset_run_url: Deprecated aliases of experiment_id / + experiment_url. The result object provides a format() method for human-readable output: ```python @@ -176,8 +179,8 @@ def accuracy_evaluator(*, input, output, expected_output=None, **kwargs): evaluators=[accuracy_evaluator] ) - print(f"Evaluated {len(result['item_results'])} questions") - print(f"View detailed results: {result['dataset_run_url']}") + print(f"Evaluated {len(result.item_results)} questions") + print(f"View detailed results: {result.experiment_url}") ``` Advanced experiment with multiple evaluators and run-level analysis: @@ -244,11 +247,11 @@ def content_diversity(*, item_results, **kwargs): ) # Results are automatically linked to dataset in Langfuse - print(f"Experiment completed! View in Langfuse: {result['dataset_run_url']}") + print(f"Experiment completed! View in Langfuse: {result.experiment_url}") # Access individual results - for i, item_result in enumerate(result["item_results"]): - print(f"Item {i+1}: {item_result['evaluations']}") + for i, item_result in enumerate(result.item_results): + print(f"Item {i+1}: {item_result.evaluations}") ``` Comparing different model versions: @@ -274,15 +277,15 @@ def content_diversity(*, item_results, **kwargs): # Both experiments are now visible in Langfuse for easy comparison print("Compare results in Langfuse:") - print(f"GPT-4: {result_gpt4.dataset_run_url}") - print(f"Custom: {result_custom.dataset_run_url}") + print(f"GPT-4: {result_gpt4.experiment_url}") + print(f"Custom: {result_custom.experiment_url}") ``` Note: - - All experiment results are automatically tracked in Langfuse as dataset runs + - All experiment results are automatically tracked in Langfuse as experiments on this dataset - Dataset items provide .input, .expected_output, and .metadata attributes - Results can be easily compared across different experiment runs in the UI - - The dataset_run_url provides direct access to detailed results and analysis + - The experiment_url provides direct access to detailed results and analysis - Failed items are handled gracefully and logged without stopping the experiment - This method works in both sync and async contexts (Jupyter notebooks, web apps, etc.) - Async execution is handled automatically with smart event loop detection diff --git a/langfuse/experiment.py b/langfuse/experiment.py index aa5481829..ee29d8611 100644 --- a/langfuse/experiment.py +++ b/langfuse/experiment.py @@ -6,6 +6,7 @@ """ import asyncio +import warnings from datetime import datetime from typing import ( TYPE_CHECKING, @@ -216,6 +217,14 @@ def __init__( self.config_id = config_id +def _warn_deprecated_alias(old: str, new: str) -> None: + warnings.warn( + f"{old} is deprecated and will be removed in a future major version. Use {new} instead.", + DeprecationWarning, + stacklevel=3, + ) + + class ExperimentItemResult: """Result structure for individual experiment items. @@ -233,8 +242,8 @@ class ExperimentItemResult: contains a name, value, optional comment, and optional metadata. trace_id: Optional Langfuse trace ID for this item's execution. Used to link the experiment result with the detailed trace in Langfuse UI. - dataset_run_id: Optional dataset run ID if this item was part of a - Langfuse dataset. None for local experiments. + experiment_id: ID of the experiment run this item belongs to. + dataset_run_id: Deprecated alias of `experiment_id`. Examples: Accessing item result data: @@ -275,7 +284,8 @@ def __init__( output: Any, evaluations: List[Evaluation], trace_id: Optional[str], - dataset_run_id: Optional[str], + experiment_id: Optional[str] = None, + dataset_run_id: Optional[str] = None, ): """Initialize an ExperimentItemResult with the provided data. @@ -284,7 +294,9 @@ def __init__( output: The actual output produced by the task function for this item. evaluations: List of evaluation results for this item. trace_id: Optional Langfuse trace ID for this item's execution. - dataset_run_id: Optional dataset run ID if this item was part of a Langfuse dataset. + experiment_id: ID of the experiment run this item belongs to. + dataset_run_id: Deprecated alias of `experiment_id`. Used only when + `experiment_id` is not provided. Note: All arguments must be provided as keywords. Positional arguments will raise a TypeError. @@ -293,7 +305,13 @@ def __init__( self.output = output self.evaluations = evaluations self.trace_id = trace_id - self.dataset_run_id = dataset_run_id + self.experiment_id = experiment_id or dataset_run_id + + @property + def dataset_run_id(self) -> Optional[str]: + """Deprecated alias of `experiment_id`.""" + _warn_deprecated_alias("ExperimentItemResult.dataset_run_id", "experiment_id") + return self.experiment_id class ExperimentResult: @@ -312,10 +330,13 @@ class ExperimentResult: run_evaluations: List of aggregate evaluation results computed across all items, such as average scores, statistical summaries, or cross-item analyses. experiment_id: ID of the experiment run propagated across all items. For - Langfuse datasets, this matches the dataset run ID. For local experiments, - this is a stable SDK-generated identifier for the run. - dataset_run_id: Optional ID of the dataset run in Langfuse (when using Langfuse datasets). - dataset_run_url: Optional direct URL to view the experiment results in Langfuse UI. + Langfuse datasets, it is derived from the project, dataset and run name, + so re-running with the same `run_name` adds to the same experiment. For + local experiments, it is a random SDK-generated identifier for the run. + experiment_url: Optional direct URL to view the experiment results in the + Langfuse UI. None if the project ID could not be resolved. + dataset_run_id: Deprecated alias of `experiment_id`. + dataset_run_url: Deprecated alias of `experiment_url`. Examples: Basic usage with local dataset: @@ -347,8 +368,8 @@ class ExperimentResult: ) # View in Langfuse UI - if result.dataset_run_url: - print(f"View detailed results: {result.dataset_run_url}") + if result.experiment_url: + print(f"View detailed results: {result.experiment_url}") ``` Formatted output: @@ -373,6 +394,7 @@ def __init__( item_results: List[ExperimentItemResult], run_evaluations: List[Evaluation], experiment_id: str, + experiment_url: Optional[str] = None, dataset_run_id: Optional[str] = None, dataset_run_url: Optional[str] = None, ): @@ -385,8 +407,11 @@ def __init__( item_results: List of results from processing individual dataset items. run_evaluations: List of aggregate evaluation results for the entire run. experiment_id: ID of the experiment run. - dataset_run_id: Optional ID of the dataset run (for Langfuse datasets). - dataset_run_url: Optional URL to view results in Langfuse UI. + experiment_url: Optional URL to view results in Langfuse UI. + dataset_run_id: Deprecated and ignored; `dataset_run_id` always + returns `experiment_id`. + dataset_run_url: Deprecated alias of `experiment_url`. Used only when + `experiment_url` is not provided. """ self.name = name self.run_name = run_name @@ -394,8 +419,19 @@ def __init__( self.item_results = item_results self.run_evaluations = run_evaluations self.experiment_id = experiment_id - self.dataset_run_id = dataset_run_id - self.dataset_run_url = dataset_run_url + self.experiment_url = experiment_url or dataset_run_url + + @property + def dataset_run_id(self) -> str: + """Deprecated alias of `experiment_id`.""" + _warn_deprecated_alias("ExperimentResult.dataset_run_id", "experiment_id") + return self.experiment_id + + @property + def dataset_run_url(self) -> Optional[str]: + """Deprecated alias of `experiment_url`.""" + _warn_deprecated_alias("ExperimentResult.dataset_run_url", "experiment_url") + return self.experiment_url def format(self, *, include_item_results: bool = False) -> str: r"""Format the experiment result for human-readable display. @@ -426,7 +462,7 @@ def format(self, *, include_item_results: bool = False) -> str: - List of all evaluation metrics that were applied - Average scores across all items for each numeric metric - Run-level evaluation results with comments - - Dataset run URL for viewing in Langfuse UI (if applicable) + - Experiment URL for viewing in Langfuse UI (if available) - Individual item details including inputs, outputs, and scores (if requested) Examples: @@ -588,9 +624,8 @@ def format(self, *, include_item_results: bool = False) -> str: output += f"\n 💭 {run_eval.comment}" output += "\n" - # Add dataset run URL if available - if self.dataset_run_url: - output += f"\n🔗 Dataset Run:\n {self.dataset_run_url}" + if self.experiment_url: + output += f"\n🔗 Experiment:\n {self.experiment_url}" return output @@ -845,7 +880,7 @@ def __call__( - output: The task function's output for this item - evaluations: List of item-level evaluation results - trace_id: Langfuse trace ID for this execution - - dataset_run_id: Dataset run ID (if using Langfuse datasets) + - experiment_id: ID of the experiment run Note: This list only includes items that were successfully processed. Failed items are excluded but logged separately. @@ -1106,7 +1141,7 @@ def __init__( dataset_version: Optional pinned dataset version. Injected by the action when ``dataset_version`` is configured. metadata: Default metadata attached to every experiment trace and - the dataset run. The action injects GitHub-sourced tags (SHA, + the experiment run. The action injects GitHub-sourced tags (SHA, PR link, workflow run link, branch, GH user, etc.). Merged with any ``metadata`` passed to :meth:`run_experiment`, with user-supplied keys winning on collision. diff --git a/tests/unit/test_experiment.py b/tests/unit/test_experiment.py index 3eb55ad40..5a2f31924 100644 --- a/tests/unit/test_experiment.py +++ b/tests/unit/test_experiment.py @@ -2,7 +2,7 @@ import inspect import typing -from datetime import datetime +from datetime import datetime, timezone from typing import get_type_hints from unittest.mock import MagicMock @@ -12,7 +12,9 @@ from langfuse import Evaluation, RegressionError, RunnerContext from langfuse._client.attributes import LangfuseOtelSpanAttributes from langfuse._client.client import Langfuse +from langfuse.api import DatasetItem, DatasetStatus from langfuse.batch_evaluation import CompositeEvaluatorFunction +from langfuse.experiment import ExperimentItemResult, ExperimentResult def _noop_task(*, item, **kwargs): # pragma: no cover - never invoked via mock @@ -524,3 +526,214 @@ def failing_evaluator(**kwargs): "failed_evaluator_count": 1, "skipped_evaluator_count": 1, } + + +def _dataset_item(*, item_id: str, dataset_id: str, input: str) -> DatasetItem: + return DatasetItem( + id=item_id, + status=DatasetStatus.ACTIVE, + input=input, + expected_output=f"expected {input}", + metadata=None, + source_trace_id=None, + source_observation_id=None, + dataset_id=dataset_id, + dataset_name="dataset", + media_references=[], + created_at=datetime(2026, 1, 1, tzinfo=timezone.utc), + updated_at=datetime(2026, 1, 1, tzinfo=timezone.utc), + ) + + +def _run_scores(create_score: MagicMock) -> list: + return [ + call.kwargs + for call in create_score.call_args_list + if call.kwargs.get("dataset_run_id") is not None + ] + + +class TestExperimentRunIdentity: + @pytest.mark.parametrize( + ("project_id", "dataset_id", "run_name", "expected"), + [ + ("p", "d", "r", "eab007015b1c6f77"), + ("proj-ü", "ds", 'Run – ✓ "q"', "8d4425c67052ced3"), + ], + ) + def test_dataset_experiment_id_matches_server_derivation( + self, langfuse_memory_client, project_id, dataset_id, run_name, expected + ): + assert ( + langfuse_memory_client._create_experiment_id( + project_id=project_id, dataset_id=dataset_id, run_name=run_name + ) + == expected + ) + + def test_dataset_experiment_without_project_id_uses_random_id( + self, langfuse_memory_client, caplog + ): + ids = { + langfuse_memory_client._create_experiment_id( + project_id=None, dataset_id="d", run_name="r" + ) + for _ in range(2) + } + + assert len(ids) == 2 + assert all(len(i) == 16 for i in ids) + assert "random experiment id" in caplog.text + + def test_dataset_run_uses_stable_id_url_version_and_no_run_item_post( + self, langfuse_memory_client, find_spans, monkeypatch + ): + create_score = MagicMock() + create_run_item = MagicMock() + monkeypatch.setattr(langfuse_memory_client, "create_score", create_score) + monkeypatch.setattr( + langfuse_memory_client.api.dataset_run_items, "create", create_run_item + ) + monkeypatch.setattr(langfuse_memory_client, "_get_project_id", lambda: "p") + version = datetime(2026, 2, 3, 4, 5, 6, tzinfo=timezone.utc) + + result = langfuse_memory_client.run_experiment( + name="exp", + run_name="r", + data=[ + _dataset_item(item_id="i1", dataset_id="d", input="a"), + _dataset_item(item_id="i2", dataset_id="d", input="b"), + ], + task=lambda *, item, **kwargs: item.input, + run_evaluators=[lambda **kwargs: Evaluation(name="run", value=1.0)], + _dataset_version=version, + ) + langfuse_memory_client.flush() + + create_run_item.assert_not_called() + assert result.experiment_id == "eab007015b1c6f77" + assert ( + result.experiment_url + == "http://test-host/project/p/experiments/results?baseline=eab007015b1c6f77" + ) + assert {r.experiment_id for r in result.item_results} == {result.experiment_id} + + spans = find_spans("experiment-item-run") + find_spans("experiment-item-task") + assert len(spans) == 4 + for span in spans: + assert ( + span.attributes[LangfuseOtelSpanAttributes.EXPERIMENT_ID] + == result.experiment_id + ) + assert ( + span.attributes[LangfuseOtelSpanAttributes.EXPERIMENT_ITEM_VERSION] + == "2026-02-03T04:05:06Z" + ) + + run_scores = _run_scores(create_score) + assert len(run_scores) == 1 + assert run_scores[0]["dataset_run_id"] == result.experiment_id + assert run_scores[0]["name"] == "run" + assert "trace_id" not in run_scores[0] + assert "observation_id" not in run_scores[0] + assert "session_id" not in run_scores[0] + + def test_local_run_persists_run_scores_without_item_version( + self, langfuse_memory_client, find_spans, monkeypatch + ): + create_score = MagicMock() + monkeypatch.setattr(langfuse_memory_client, "create_score", create_score) + monkeypatch.setattr(langfuse_memory_client, "_get_project_id", lambda: "p") + + result = langfuse_memory_client.run_experiment( + name="exp", + run_name="r", + data=[{"input": "a"}], + task=lambda *, item, **kwargs: item["input"], + run_evaluators=[lambda **kwargs: Evaluation(name="run", value=1.0)], + _dataset_version=datetime(2026, 2, 3, tzinfo=timezone.utc), + ) + langfuse_memory_client.flush() + + assert len(result.experiment_id) == 16 + assert result.experiment_id != "eab007015b1c6f77" + assert result.experiment_url == ( + f"http://test-host/project/p/experiments/results?baseline={result.experiment_id}" + ) + assert [s["dataset_run_id"] for s in _run_scores(create_score)] == [ + result.experiment_id + ] + for span in find_spans("experiment-item-run") + find_spans( + "experiment-item-task" + ): + assert LangfuseOtelSpanAttributes.EXPERIMENT_ITEM_VERSION not in ( + span.attributes or {} + ) + + def test_unresolvable_project_id_yields_no_url( + self, langfuse_memory_client, monkeypatch + ): + def fail(): + raise RuntimeError("no network") + + monkeypatch.setattr(langfuse_memory_client, "create_score", MagicMock()) + monkeypatch.setattr(langfuse_memory_client, "_get_project_id", fail) + + result = langfuse_memory_client.run_experiment( + name="exp", + data=[_dataset_item(item_id="i1", dataset_id="d", input="a")], + task=lambda *, item, **kwargs: item.input, + ) + + assert len(result.item_results) == 1 + assert len(result.experiment_id) == 16 + assert result.experiment_url is None + + +class TestExperimentResultAliases: + def test_dataset_run_fields_are_deprecated_aliases(self): + item_result = ExperimentItemResult( + item={"input": "a"}, + output="a", + evaluations=[], + trace_id="t", + experiment_id="e", + ) + result = ExperimentResult( + name="exp", + run_name="r", + description=None, + item_results=[item_result], + run_evaluations=[], + experiment_id="e", + experiment_url="http://host/project/p/experiments/results?baseline=e", + ) + + with pytest.warns(DeprecationWarning, match="experiment_id"): + assert result.dataset_run_id == "e" + with pytest.warns(DeprecationWarning, match="experiment_url"): + assert result.dataset_run_url == result.experiment_url + with pytest.warns(DeprecationWarning, match="experiment_id"): + assert item_result.dataset_run_id == "e" + assert f"Experiment:\n {result.experiment_url}" in result.format() + + def test_legacy_constructor_arguments_still_populate_new_fields(self): + item_result = ExperimentItemResult( + item={"input": "a"}, + output="a", + evaluations=[], + trace_id="t", + dataset_run_id="legacy", + ) + result = ExperimentResult( + name="exp", + run_name="r", + description=None, + item_results=[item_result], + run_evaluations=[], + experiment_id="e", + dataset_run_url="http://legacy", + ) + + assert item_result.experiment_id == "legacy" + assert result.experiment_url == "http://legacy" diff --git a/tests/unit/test_propagate_attributes.py b/tests/unit/test_propagate_attributes.py index c49a4e956..019ceb8f0 100644 --- a/tests/unit/test_propagate_attributes.py +++ b/tests/unit/test_propagate_attributes.py @@ -2813,28 +2813,17 @@ def test_experiment_attributes_propagate_with_dataset( self, langfuse_client, memory_exporter, monkeypatch ): """Test experiment attribute propagation with Langfuse dataset.""" - created_run_items = [] - - # Mock the sync API used by run_experiment to create dataset run items - def mock_create_dataset_run_item(*args, **kwargs): - from langfuse.api import DatasetRunItem - - created_run_items.append(kwargs) - return DatasetRunItem( - id="mock-run-item-id", - dataset_run_id="mock-dataset-run-id-123", - dataset_run_name=kwargs.get("run_name", "Dataset Test"), - dataset_item_id=kwargs.get("dataset_item_id", "mock-item-id"), - trace_id="mock-trace-id", - observation_id=kwargs.get("observation_id"), - created_at=datetime.now(), - updated_at=datetime.now(), - ) + + def fail_create_dataset_run_item(*args, **kwargs): + raise AssertionError("run_experiment must not create dataset run items") monkeypatch.setattr( langfuse_client.api.dataset_run_items, "create", - mock_create_dataset_run_item, + fail_create_dataset_run_item, + ) + monkeypatch.setattr( + langfuse_client, "_get_project_id", lambda: "test-project-id" ) # Create a mock dataset with items @@ -2899,9 +2888,16 @@ def task_with_children(*, item, **kwargs): assert len(root_spans) >= 1, "Should have at least 1 root span" first_root = root_spans[0] task_span = self.get_span_by_name(memory_exporter, "experiment-item-task") - assert result.experiment_id == "mock-dataset-run-id-123" - assert len(created_run_items) == 1 - assert created_run_items[0]["observation_id"] == task_span["span_id"] + assert result.experiment_id == langfuse_client._create_experiment_id( + project_id="test-project-id", + dataset_id=dataset_id, + run_name=result.run_name, + ) + self.verify_span_attribute( + first_root, + LangfuseOtelSpanAttributes.EXPERIMENT_ITEM_ROOT_OBSERVATION_ID, + task_span["span_id"], + ) # Root-only attributes should be on root self.verify_span_attribute( From 8a567e8619813c5fb5aa7814ce2b731d13bd297b Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 6 Oct 2026 14:26:09 +0000 Subject: [PATCH 08/25] test(experiments): read experiment e2e results via v4 APIs Replace dataset run, dataset run item and legacy trace reads with the experiments, observations and v3 scores APIs, and cover stable experiment ids, run-level scores for local data and the recorded item version. Co-authored-by: Hassieb Pakzad --- tests/e2e/test_experiments.py | 435 ++++++++++++++++++++-------------- 1 file changed, 262 insertions(+), 173 deletions(-) diff --git a/tests/e2e/test_experiments.py b/tests/e2e/test_experiments.py index 80881f144..ceddba7be 100644 --- a/tests/e2e/test_experiments.py +++ b/tests/e2e/test_experiments.py @@ -1,20 +1,121 @@ """Comprehensive tests for Langfuse experiment functionality matching JS SDK.""" +import os import time -from typing import Any, Dict, List +from datetime import datetime, timedelta, timezone +from typing import Any, Callable, Dict, List, TypeVar +from uuid import uuid4 import pytest from opentelemetry import trace as otel_trace_api from langfuse import get_client from langfuse._client.attributes import LangfuseOtelSpanAttributes +from langfuse.api import LangfuseAPI from langfuse.experiment import ( Evaluation, ExperimentData, ExperimentItem, ExperimentItemResult, ) -from tests.support.utils import create_uuid, get_api, wait_for_trace + +T = TypeVar("T") + +READ_TIMEOUT_SECONDS = float(os.environ.get("LANGFUSE_E2E_READ_TIMEOUT_SECONDS", "30")) +READ_INTERVAL_SECONDS = float( + os.environ.get("LANGFUSE_E2E_READ_INTERVAL_SECONDS", "0.5") +) +ITEM_FIELDS = "core,dataset,io,metadata,experimentMetadata,scores" + + +def create_uuid() -> str: + return str(uuid4()) + + +def get_api() -> LangfuseAPI: + return LangfuseAPI( + username=os.environ.get("LANGFUSE_PUBLIC_KEY"), + password=os.environ.get("LANGFUSE_SECRET_KEY"), + base_url=os.environ.get("LANGFUSE_BASE_URL"), + ) + + +def _read_window_start() -> datetime: + return datetime.now(timezone.utc) - timedelta(hours=1) + + +def poll(operation: Callable[[], T], is_ready: Callable[[T], bool]) -> T: + """Re-read until ready or timeout; returns the last result for the caller to assert.""" + deadline = time.monotonic() + READ_TIMEOUT_SECONDS + while True: + result = operation() + if is_ready(result) or time.monotonic() >= deadline: + return result + time.sleep(READ_INTERVAL_SECONDS) + + +def get_experiment_items( + experiment_id: str, *, expected_count: int, with_scores: bool = False +) -> list: + api = get_api() + return poll( + lambda: ( + api.experiments.list_items( + from_start_time=_read_window_start(), + experiment_id=experiment_id, + fields=ITEM_FIELDS, + limit=100, + ).data + ), + lambda items: ( + len(items) >= expected_count + and (not with_scores or all(item.scores for item in items)) + ), + ) + + +def get_experiment( + experiment_id: str, *, is_ready: Callable[[Any], bool] = lambda _: True +): + api = get_api() + experiments = poll( + lambda: ( + api.experiments.list( + from_start_time=_read_window_start(), + id=experiment_id, + fields="core,metadata,scores", + ).data + ), + lambda data: len(data) == 1 and is_ready(data[0]), + ) + assert len(experiments) == 1, f"Experiment {experiment_id} should exist" + return experiments[0] + + +def get_observations(trace_id: str, *, is_ready: Callable[[list], bool]) -> list: + api = get_api() + return poll( + lambda: ( + api.observations.get_many( + trace_id=trace_id, + fields="core,basic,io,metadata", + from_start_time=_read_window_start(), + ).data + ), + is_ready, + ) + + +def get_trace_scores(trace_id: str, *, expected_count: int) -> list: + api = get_api() + return poll( + lambda: api.scores_v3.get_many_v3(trace_id=trace_id, fields="subject").data, + lambda scores: len(scores) >= expected_count, + ) + + +def has_score(name: str) -> Callable[[Any], bool]: + return lambda experiment: any(s.name == name for s in experiment.scores or []) @pytest.fixture @@ -78,63 +179,53 @@ def test_run_experiment_on_local_dataset(sample_dataset): assert len(result.item_results) == 3 assert len(result.run_evaluations) == 1 assert result.run_evaluations[0].name == "average_length" - assert result.dataset_run_id is None # No dataset_run_id for local datasets + assert len(result.experiment_id) == 16 + assert result.experiment_url is not None + assert result.experiment_url.endswith( + f"/experiments/results?baseline={result.experiment_id}" + ) + with pytest.warns(DeprecationWarning): + assert result.dataset_run_id == result.experiment_id # Validate item results structure for item_result in result.item_results: assert hasattr(item_result, "output") assert hasattr(item_result, "evaluations") assert hasattr(item_result, "trace_id") - assert ( - item_result.dataset_run_id is None - ) # No dataset_run_id for local datasets + assert item_result.experiment_id == result.experiment_id assert len(item_result.evaluations) == 2 # Both evaluators should run - # Flush and wait for server processing langfuse_client.flush() - time.sleep(2) - - # Validate traces are correctly persisted with input/output/metadata - api = get_api() - expected_inputs = ["Germany", "France", "Spain"] - expected_outputs = ["Capital of Germany", "Capital of France", "Capital of Spain"] - - for i, item_result in enumerate(result.item_results): - trace_id = item_result.trace_id - assert trace_id is not None, f"Item {i} should have a trace_id" - - # Fetch trace from API - trace = api.trace.get(trace_id) - assert trace is not None, f"Trace {trace_id} should exist" - - # Validate trace name - assert trace.name == "experiment-item-run", ( - f"Trace {trace_id} should have correct name" - ) - - # Validate trace input - should contain the experiment item - assert trace.input is not None, f"Trace {trace_id} should have input" - expected_input = expected_inputs[i] - # The input should contain the item data in some form - assert expected_input in str(trace.input), ( - f"Trace {trace_id} input should contain '{expected_input}'" - ) - # Validate trace output - should be the task result - assert trace.output is not None, f"Trace {trace_id} should have output" - expected_output = expected_outputs[i] - assert trace.output == expected_output, ( - f"Trace {trace_id} output should be '{expected_output}', got '{trace.output}'" - ) + expected = { + "Germany": ("Capital of Germany", "Berlin"), + "France": ("Capital of France", "Paris"), + "Spain": ("Capital of Spain", "Madrid"), + } + items = get_experiment_items(result.experiment_id, expected_count=3) + assert len(items) == 3 + assert {item.trace_id for item in items} == { + r.trace_id for r in result.item_results + } - # Validate trace metadata contains experiment name - assert trace.metadata is not None, f"Trace {trace_id} should have metadata" - assert "experiment_name" in trace.metadata, ( - f"Trace {trace_id} metadata should contain experiment_name" - ) - assert trace.metadata["experiment_name"] == "Euro capitals", ( - f"Trace {trace_id} metadata should have correct experiment_name" - ) + for item in items: + assert item.experiment_name == result.run_name + assert item.experiment_dataset_id is None + assert item.input in expected + expected_output, expected_answer = expected[item.input] + assert item.output == expected_output + assert item.expected_output == expected_answer + assert item.metadata is not None + assert item.metadata["experiment_name"] == "Euro capitals" + + # Run-level evaluations are persisted for local data, too + experiment = get_experiment( + result.experiment_id, is_ready=has_score("average_length") + ) + assert experiment.name == result.run_name + assert experiment.description == "Country capital experiment" + assert experiment.item_count == 3 + assert [s.name for s in experiment.scores or []] == ["average_length"] def test_run_experiment_flattens_large_metadata_for_server_ingestion(): @@ -168,21 +259,25 @@ def task_with_external_child(*, item: ExperimentItem, **kwargs: Dict[str, Any]): trace_id = result.item_results[0].trace_id assert trace_id is not None - trace = wait_for_trace( + observations = get_observations( trace_id, - is_result_ready=lambda fetched_trace: any( - observation.name == external_span_name - for observation in fetched_trace.observations + is_ready=lambda observations: any( + observation.name == external_span_name for observation in observations ), ) - assert trace.metadata is not None + root_observation = next( + observation + for observation in observations + if observation.name == "experiment-item-run" + ) + assert root_observation.metadata is not None for metadata_key, metadata_value in experiment_metadata.items(): - assert trace.metadata[metadata_key] == metadata_value + assert root_observation.metadata[metadata_key] == metadata_value external_observation = next( observation - for observation in trace.observations + for observation in observations if observation.name == external_span_name ) external_metadata = external_observation.metadata or {} @@ -227,117 +322,109 @@ def test_run_experiment_on_langfuse_dataset(): run_evaluators=[run_evaluator_average_length], ) - # Should have dataset run ID for Langfuse datasets - assert result.dataset_run_id is not None + project_id = langfuse_client._get_project_id() + assert project_id is not None + assert result.experiment_id == langfuse_client._create_experiment_id( + project_id=project_id, dataset_id=dataset.id, run_name=result.run_name + ) + assert result.experiment_url == ( + f"{os.environ['LANGFUSE_BASE_URL']}/project/{project_id}" + f"/experiments/results?baseline={result.experiment_id}" + ) assert len(result.item_results) == 2 - assert all(item.dataset_run_id is not None for item in result.item_results) + assert all( + item.experiment_id == result.experiment_id for item in result.item_results + ) - # Flush and wait for server processing langfuse_client.flush() - time.sleep(3) - - # Verify dataset run exists via API - api = get_api() - dataset_run = api.datasets.get_run( - dataset_name=dataset_name, run_name=result.run_name - ) - # Validate traces are correctly persisted with input/output/metadata expected_data = {"Germany": "Capital of Germany", "France": "Capital of France"} - dataset_run_id = result.dataset_run_id - - # Create a mapping from dataset item ID to dataset item for validation dataset_item_map = {item.id: item for item in dataset.items} - for i, item_result in enumerate(result.item_results): - trace_id = item_result.trace_id - assert trace_id is not None, f"Item {i} should have a trace_id" + items = get_experiment_items( + result.experiment_id, expected_count=2, with_scores=True + ) + assert len(items) == 2, "Experiment should have 2 items" + assert {item.trace_id for item in items} == { + r.trace_id for r in result.item_results + } + assert {item.experiment_item_id for item in items} == set(dataset_item_map) - # Fetch trace from API - trace = api.trace.get(trace_id) - assert trace is not None, f"Trace {trace_id} should exist" + for item in items: + assert item.experiment_name == result.run_name + assert item.experiment_dataset_id == dataset.id + assert item.experiment_description == "Test on Langfuse dataset" + assert item.experiment_item_version is None - # Validate trace name - assert trace.name == "experiment-item-run", ( - f"Trace {trace_id} should have correct name" - ) + dataset_item = dataset_item_map[item.experiment_item_id] + assert item.input == dataset_item.input + assert item.output == expected_data[dataset_item.input] + assert item.expected_output == dataset_item.expected_output - # Validate trace input and output match expected pairs - assert trace.input is not None, f"Trace {trace_id} should have input" - trace_input_str = str(trace.input) + assert item.metadata is not None + assert item.metadata["experiment_name"] == experiment_name + assert item.metadata["dataset_id"] == dataset.id + assert item.metadata["dataset_item_id"] == item.experiment_item_id - # Find which expected input this trace corresponds to - matching_input = None - for expected_input in expected_data.keys(): - if expected_input in trace_input_str: - matching_input = expected_input - break + assert [s.name for s in item.scores or []] == ["factuality"] - assert matching_input is not None, ( - f"Trace {trace_id} input '{trace_input_str}' should contain one of {list(expected_data.keys())}" - ) + experiment = get_experiment( + result.experiment_id, is_ready=has_score("average_length") + ) + assert experiment.name == result.run_name + assert experiment.description == "Test on Langfuse dataset" + assert experiment.dataset_id == dataset.id + assert experiment.item_count == 2 + assert [s.name for s in experiment.scores or []] == ["average_length"] + assert experiment.scores[0].subject.kind == "experiment" + assert experiment.scores[0].subject.id == result.experiment_id - # Validate trace output matches the expected output for this input - assert trace.output is not None, f"Trace {trace_id} should have output" - expected_output = expected_data[matching_input] - assert trace.output == expected_output, ( - f"Trace {trace_id} output should be '{expected_output}', got '{trace.output}'" - ) - # Validate trace metadata contains experiment and dataset info - assert trace.metadata is not None, f"Trace {trace_id} should have metadata" - assert "experiment_name" in trace.metadata, ( - f"Trace {trace_id} metadata should contain experiment_name" - ) - assert trace.metadata["experiment_name"] == experiment_name, ( - f"Trace {trace_id} metadata should have correct experiment_name" - ) +def test_run_experiment_on_versioned_dataset_records_item_version(): + """Pinned dataset versions are recorded as the experiment item version.""" + langfuse_client = get_client() + dataset_name = "versioned-dataset-" + create_uuid() + langfuse_client.create_dataset(name=dataset_name) + langfuse_client.create_dataset_item( + dataset_name=dataset_name, input="Germany", expected_output="Berlin" + ) - # Validate dataset-specific metadata fields - assert "dataset_id" in trace.metadata, ( - f"Trace {trace_id} metadata should contain dataset_id" - ) - assert trace.metadata["dataset_id"] == dataset.id, ( - f"Trace {trace_id} metadata should have correct dataset_id" - ) + version = datetime.now(timezone.utc).replace(microsecond=0) + timedelta(seconds=1) + time.sleep(1.5) + dataset = langfuse_client.get_dataset(dataset_name, version=version) + assert len(dataset.items) == 1 - assert "dataset_item_id" in trace.metadata, ( - f"Trace {trace_id} metadata should contain dataset_item_id" - ) - # Get the dataset item ID from metadata and validate it exists - dataset_item_id = trace.metadata["dataset_item_id"] - assert dataset_item_id in dataset_item_map, ( - f"Trace {trace_id} metadata dataset_item_id should correspond to a valid dataset item" - ) + result = dataset.run_experiment( + name="Versioned " + create_uuid()[:8], task=mock_task + ) + langfuse_client.flush() - # Validate the dataset item input matches the trace input - dataset_item = dataset_item_map[dataset_item_id] - assert dataset_item.input == matching_input, ( - f"Trace {trace_id} should correspond to dataset item with input '{matching_input}'" - ) + items = get_experiment_items(result.experiment_id, expected_count=1) + assert len(items) == 1 + assert items[0].experiment_item_version is not None + assert items[0].experiment_item_version.astimezone(timezone.utc) == version - assert dataset_run is not None, f"Dataset run {dataset_run_id} should exist" - assert dataset_run.name == result.run_name, "Dataset run should have correct name" - assert dataset_run.description == "Test on Langfuse dataset", ( - "Dataset run should have correct description" - ) - # Get dataset run items to verify trace linkage - dataset_run_items = api.dataset_run_items.list( - dataset_id=dataset.id, run_name=result.run_name +def test_same_run_name_on_dataset_reuses_experiment_id(): + """Re-running with the same run name on a dataset groups into one experiment.""" + langfuse_client = get_client() + dataset_name = "rerun-dataset-" + create_uuid() + langfuse_client.create_dataset(name=dataset_name) + langfuse_client.create_dataset_item( + dataset_name=dataset_name, input="Germany", expected_output="Berlin" ) - assert len(dataset_run_items.data) == 2, "Dataset run should have 2 items" + dataset = langfuse_client.get_dataset(dataset_name) + run_name = "Shared run " + create_uuid()[:8] - # Verify each dataset run item links to the correct trace - run_item_trace_ids = { - item.trace_id for item in dataset_run_items.data if item.trace_id - } - result_trace_ids = {item.trace_id for item in result.item_results} + first = dataset.run_experiment(name="Rerun", run_name=run_name, task=mock_task) + second = dataset.run_experiment(name="Rerun", run_name=run_name, task=mock_task) + langfuse_client.flush() - assert run_item_trace_ids == result_trace_ids, ( - f"Dataset run items should link to the same traces as experiment results. " - f"Run items: {run_item_trace_ids}, Results: {result_trace_ids}" - ) + assert first.experiment_id == second.experiment_id + items = get_experiment_items(first.experiment_id, expected_count=2) + assert {item.trace_id for item in items} == { + r.trace_id for r in first.item_results + second.item_results + } # Error Handling Tests @@ -630,20 +717,21 @@ def test_run_evaluator(**kwargs): run_evaluators=[test_run_evaluator], ) - assert result.dataset_run_id is not None assert len(result.item_results) == 1 assert len(result.run_evaluations) == 1 langfuse_client.flush() - time.sleep(3) - # Verify scores are persisted via API - api = get_api() - dataset_run = api.datasets.get_run( - dataset_name=dataset_name, run_name=result.run_name + experiment = get_experiment( + result.experiment_id, is_ready=has_score("persistence_run_test") ) + assert experiment.name == "Score persistence test" + run_scores = {s.name: s for s in experiment.scores or []} + assert run_scores["persistence_run_test"].value == 0.9 - assert dataset_run.name == "Score persistence test" + trace_scores = get_trace_scores(result.item_results[0].trace_id, expected_count=1) + assert [(s.name, s.value) for s in trace_scores] == [("persistence_test", 0.85)] + assert trace_scores[0].subject.kind == "observation" def test_multiple_experiments_on_same_dataset(): @@ -688,21 +776,22 @@ def test_multiple_experiments_on_same_dataset(): ) langfuse_client.flush() - time.sleep(2) - # Both experiments should have different run IDs - assert result1.dataset_run_id is not None - assert result2.dataset_run_id is not None - assert result1.dataset_run_id != result2.dataset_run_id + assert result1.experiment_id != result2.experiment_id - # Verify both runs exist in database api = get_api() - runs = api.datasets.get_runs(dataset_name) - assert len(runs.data) >= 2 - - run_names = [run.name for run in runs.data] - assert "Experiment 1" in run_names - assert "Experiment 2" in run_names + experiments = poll( + lambda: ( + api.experiments.list( + from_start_time=_read_window_start(), dataset_id=dataset.id + ).data + ), + lambda data: len(data) >= 2, + ) + assert {e.id: e.name for e in experiments} == { + result1.experiment_id: "Experiment 1", + result2.experiment_id: "Experiment 2", + } # Result Formatting Tests @@ -840,24 +929,24 @@ def mock_task_with_boolean_results(*, item: ExperimentItem, **kwargs): assert run_eval.value is False # Spain should fail, so not all pass assert run_eval.data_type == ScoreDataType.BOOLEAN - # Flush and wait for server processing langfuse_client.flush() - time.sleep(3) # Verify scores are persisted via API with correct data types for i, item_result in enumerate(result.item_results): trace_id = item_result.trace_id assert trace_id is not None, f"Item {i} should have a trace_id" - # Fetch trace from API to verify score persistence - trace = wait_for_trace( - trace_id, - is_result_ready=lambda trace: len(trace.scores) > 0, - ) - assert trace is not None, f"Trace {trace_id} should exist" + scores = get_trace_scores(trace_id, expected_count=1) + assert len(scores) == 1 + assert scores[0].data_type == "BOOLEAN" + assert scores[0].value is expected_results[i] - for score in trace.scores: - assert score.data_type == "BOOLEAN" + experiment = get_experiment( + result.experiment_id, is_ready=has_score("all_items_pass") + ) + run_score = next(s for s in experiment.scores or [] if s.name == "all_items_pass") + assert run_score.data_type == "BOOLEAN" + assert run_score.value is False def test_experiment_composite_evaluator_weighted_average(): From a5c964ea016f48adc813cd5b81e97bb673b4134f Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 6 Oct 2026 14:29:29 +0000 Subject: [PATCH 09/25] test(experiments): pin the cross-SDK experiment id vectors Co-authored-by: Hassieb Pakzad --- tests/unit/test_experiment.py | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/tests/unit/test_experiment.py b/tests/unit/test_experiment.py index 5a2f31924..5f279f31d 100644 --- a/tests/unit/test_experiment.py +++ b/tests/unit/test_experiment.py @@ -559,6 +559,19 @@ class TestExperimentRunIdentity: [ ("p", "d", "r", "eab007015b1c6f77"), ("proj-ü", "ds", 'Run – ✓ "q"', "8d4425c67052ced3"), + # Shared with the JS SDK and the platform; keep these in sync. + ( + "7a88fb47-b4e2-43b8-a06c-a5ce950dc53a", + "cm9x1dataset0000000000001", + "my-run", + "a1164c0f238e4173", + ), + ( + "7a88fb47-b4e2-43b8-a06c-a5ce950dc53a", + "cm9x1dataset0000000000001", + 'Läufe "v2" 🚀 – 2026-10-06T12:00:00.000Z', + "dce21641128b88a8", + ), ], ) def test_dataset_experiment_id_matches_server_derivation( From dfb67a58720b4c3e7bde5d0ddf6949e598622dee Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 6 Oct 2026 14:22:40 +0000 Subject: [PATCH 10/25] feat(batch-evaluation)!: evaluate observations via the v2 observations API Batch evaluation now reads exclusively from GET /api/public/v2/observations with cursor pagination, so it works on Langfuse platform v4. BREAKING CHANGES: - scope="traces" is replaced by scope="root_observations", which evaluates the logical root observation of each trace and attaches scores to the trace. scope="observations" is unchanged in intent. - Mappers receive an ObservationV2. input/output are raw strings and only the requested field groups are populated. - fetch_trace_fields is replaced by fields (v2 field groups), defaulting to "core,basic,io,metadata" in both the client and the runner. - filter must be a JSON array in the v2 observations filter format. - Resume tokens carry the pagination cursor and are also returned when max_items is reached while more items exist. - The private _additional_trace_tags option is removed. Co-authored-by: Hassieb Pakzad --- langfuse/_client/client.py | 171 +++--- langfuse/batch_evaluation.py | 830 +++++++++++----------------- tests/unit/test_batch_evaluation.py | 376 +++++++++++++ 3 files changed, 787 insertions(+), 590 deletions(-) create mode 100644 tests/unit/test_batch_evaluation.py diff --git a/langfuse/_client/client.py b/langfuse/_client/client.py index 653243b7d..5d59cfea2 100644 --- a/langfuse/_client/client.py +++ b/langfuse/_client/client.py @@ -3178,11 +3178,11 @@ def _create_experiment_run_name( def run_batched_evaluation( self, *, - scope: Literal["traces", "observations"], + scope: Literal["observations", "root_observations"], mapper: MapperFunction, filter: Optional[str] = None, fetch_batch_size: int = 50, - fetch_trace_fields: Optional[str] = None, + fields: Optional[str] = "core,basic,io,metadata", max_items: Optional[int] = None, max_retries: int = 3, evaluators: List[EvaluatorFunction], @@ -3190,16 +3190,15 @@ def run_batched_evaluation( max_concurrency: int = 5, metadata: Optional[Dict[str, Any]] = None, _add_observation_scores_to_trace: bool = False, - _additional_trace_tags: Optional[List[str]] = None, resume_from: Optional[BatchEvaluationResumeToken] = None, verbose: bool = False, ) -> BatchEvaluationResult: - """Fetch traces or observations using legacy read APIs and evaluate each item. + """Fetch observations from Langfuse and evaluate each of them. This method provides a powerful way to evaluate existing data in Langfuse at scale. - It fetches items based on filters, transforms them using a mapper function, runs - evaluators on each item, and creates scores that are linked back to the original - entities. This is ideal for: + It fetches observations based on filters, transforms them using a mapper function, + runs evaluators on each item, and creates scores that are linked back to the + original entities. This is ideal for: - Running evaluations on production traces after deployment - Backtesting new evaluation metrics on historical data @@ -3210,46 +3209,55 @@ def run_batched_evaluation( it memory-efficient for large datasets. It includes comprehensive error handling, retry logic, and resume capability for long-running evaluations. - Legacy platform compatibility: - This method reads traces from `GET /api/public/traces` and observations - from the legacy `GET /api/public/observations` endpoint. It is supported - with Langfuse platform v3 and is not yet supported with platform v4. + Items are read from `GET /api/public/v2/observations` with cursor pagination, + newest first (by start time). Args: - scope: The type of items to evaluate. Must be one of: - - "traces": Evaluate complete traces with all their observations - - "observations": Evaluate individual observations (spans, generations, events) - mapper: Function that transforms API response objects into evaluator inputs. - Receives a trace/observation object and returns an EvaluatorInputs + scope: Which observations to evaluate. Must be one of: + - "observations": Every observation matching the filter (spans, + generations, events, ...). Scores are attached to the observation. + - "root_observations": Only the logical root observation of each trace. + Scores are attached to the trace. Use this to evaluate whole traces. + mapper: Function that transforms an `ObservationV2` into evaluator inputs. + Called as `mapper(item=observation)` and must return an EvaluatorInputs instance with input, output, expected_output, and metadata fields. - Can be sync or async. + `input`/`output` are raw strings (not JSON-parsed), and only the + requested `fields` groups are populated. Can be sync or async. evaluators: List of evaluation functions to run on each item. Each evaluator receives the mapped inputs and returns Evaluation object(s). Evaluator failures are logged but don't stop the batch evaluation. - filter: Optional JSON filter string for querying items (same format as Langfuse API). Examples: - - '{"tags": ["production"]}' - - '{"user_id": "user123", "timestamp": {"operator": ">", "value": "2024-01-01"}}' - Default: None (fetches all items). + filter: Optional JSON array of filter conditions in the v2 observations + filter format, for example: + - '[{"type": "arrayOptions", "column": "tags", "operator": "any of", "value": ["production"]}]' + - '[{"type": "string", "column": "traceName", "operator": "=", "value": "chat"}]' + - '[{"type": "datetime", "column": "startTime", "operator": ">=", "value": "2026-01-01T00:00:00Z"}]' + Default: None (fetches all items of the scope). fetch_batch_size: Number of items to fetch per API call and hold in memory. - Larger values may be faster but use more memory. Default: 50. - fetch_trace_fields: Comma-separated list of fields to include when fetching traces. Available field groups: 'core' (always included), 'io' (input, output, metadata), 'scores', 'observations', 'metrics'. If not specified, all fields are returned. Example: 'core,scores,metrics'. Note: Excluded 'observations' or 'scores' fields return empty arrays; excluded 'metrics' returns -1 for 'totalCost' and 'latency'. Only relevant if scope is 'traces'. + Larger values may be faster but use more memory. Maximum 1000. Default: 50. + fields: Comma-separated list of observation field groups to fetch. Available + groups: 'core' (always included), 'basic', 'time', 'io', 'metadata', + 'model', 'usage', 'prompt', 'metrics', 'trace_context'. Fields of groups + that are not requested are None on the item passed to the mapper. + Metadata values longer than 200 characters are truncated. + Default: "core,basic,io,metadata". max_items: Maximum total number of items to process. If None, processes all items matching the filter. Useful for testing or limiting evaluation runs. Default: None (process all). max_concurrency: Maximum number of items to evaluate concurrently. Controls parallelism and resource usage. Default: 5. composite_evaluator: Optional function that creates a composite score from - item-level evaluations. Receives the original item and its evaluations, - returns a single Evaluation. Useful for weighted averages or combined metrics. - Default: None. + item-level evaluations. Receives the mapped inputs and the item's + evaluations, returns Evaluation(s). Useful for weighted averages or + combined metrics. Default: None. metadata: Optional metadata dict to add to all created scores. Useful for tracking evaluation runs, versions, or other context. Default: None. max_retries: Maximum number of retry attempts for failed batch fetches. - Uses exponential backoff (1s, 2s, 4s). Default: 3. + Default: 3. verbose: If True, logs progress information to console. Useful for monitoring long-running evaluations. Default: False. - resume_from: Optional resume token from a previous incomplete run. Allows - continuing evaluation after interruption or failure. Default: None. + resume_from: Optional resume token from a previous run that stopped early + (fetch failure or `max_items`). Continues exactly after the last + processed page. Pass the same `scope` and `filter`. Default: None. Returns: @@ -3261,45 +3269,45 @@ def run_batched_evaluation( - total_composite_scores_created: Scores created by composite evaluator - total_evaluations_failed: Individual evaluator failures - evaluator_stats: Per-evaluator statistics (success rate, scores created) - - resume_token: Token for resuming if incomplete (None if completed) - - completed: True if all items processed + - resume_token: Token for continuing the run (set after a fetch failure + or when max_items was reached while more items exist) + - completed: False if the run stopped because a fetch failed - duration_seconds: Total execution time - - failed_item_ids: IDs of items that failed + - failed_item_ids: Observation IDs of items that failed - error_summary: Error types and counts - has_more_items: True if max_items reached but more exist + - item_evaluations: Evaluations per observation ID Raises: - ValueError: If invalid scope is provided. + ValueError: If an invalid scope or a non-array filter is provided, or the + resume token belongs to a different scope. Examples: - Basic trace evaluation: + Evaluate whole traces via their root observations: ```python from langfuse import Langfuse, EvaluatorInputs, Evaluation client = Langfuse() - # Define mapper to extract fields from traces - def trace_mapper(trace): + def root_mapper(*, item): return EvaluatorInputs( - input=trace.input, - output=trace.output, + input=item.input, + output=item.output, expected_output=None, - metadata={"trace_id": trace.id} + metadata={"trace_id": item.trace_id}, ) - # Define evaluator def length_evaluator(*, input, output, expected_output, metadata): return Evaluation( name="output_length", value=len(output) if output else 0 ) - # Run batch evaluation result = client.run_batched_evaluation( - scope="traces", - mapper=trace_mapper, + scope="root_observations", + mapper=root_mapper, evaluators=[length_evaluator], - filter='{"tags": ["production"]}', + filter='[{"type": "arrayOptions", "column": "tags", "operator": "any of", "value": ["production"]}]', max_items=1000, verbose=True ) @@ -3308,87 +3316,69 @@ def length_evaluator(*, input, output, expected_output, metadata): print(f"Created {result.total_scores_created} scores") ``` - Evaluation with composite scorer: + Evaluate generations with a composite scorer: ```python + import json + + def generation_mapper(*, item): + return EvaluatorInputs( + input=json.loads(item.input) if item.input else None, + output=item.output, + expected_output=None, + metadata={"model": item.model}, + ) + def accuracy_evaluator(*, input, output, expected_output, metadata): - # ... evaluation logic return Evaluation(name="accuracy", value=0.85) def relevance_evaluator(*, input, output, expected_output, metadata): - # ... evaluation logic return Evaluation(name="relevance", value=0.92) - def composite_evaluator(*, item, evaluations): - # Weighted average of evaluations + def composite_evaluator(*, input, output, expected_output, metadata, evaluations): weights = {"accuracy": 0.6, "relevance": 0.4} total = sum( e.value * weights.get(e.name, 0) for e in evaluations if isinstance(e.value, (int, float)) ) - return Evaluation( - name="composite_score", - value=total, - comment=f"Weighted average of {len(evaluations)} metrics" - ) + return Evaluation(name="composite_score", value=total) result = client.run_batched_evaluation( - scope="traces", - mapper=trace_mapper, + scope="observations", + mapper=generation_mapper, evaluators=[accuracy_evaluator, relevance_evaluator], composite_evaluator=composite_evaluator, - filter='{"user_id": "important_user"}', - verbose=True + filter='[{"type": "string", "column": "type", "operator": "=", "value": "GENERATION"}]', + fields="core,basic,io,model", ) ``` - Handling incomplete runs with resume: + Continuing a run that stopped early: ```python - # Initial run that may fail or timeout result = client.run_batched_evaluation( scope="observations", - mapper=obs_mapper, - evaluators=[my_evaluator], + mapper=generation_mapper, + evaluators=[accuracy_evaluator], max_items=10000, - verbose=True ) - # Check if incomplete - if not result.completed and result.resume_token: - print(f"Processed {result.resume_token.items_processed} items before interruption") - - # Resume from where it left off + while result.resume_token: result = client.run_batched_evaluation( scope="observations", - mapper=obs_mapper, - evaluators=[my_evaluator], + mapper=generation_mapper, + evaluators=[accuracy_evaluator], + max_items=10000, resume_from=result.resume_token, - verbose=True ) - - print(f"Total items processed: {result.total_items_processed}") - ``` - - Monitoring evaluator performance: - ```python - result = client.run_batched_evaluation(...) - - for stats in result.evaluator_stats: - success_rate = stats.successful_runs / stats.total_runs - print(f"{stats.name}:") - print(f" Success rate: {success_rate:.1%}") - print(f" Scores created: {stats.total_scores_created}") - - if stats.failed_runs > 0: - print(f" ⚠️ Failed {stats.failed_runs} times") ``` Note: - Evaluator failures are logged but don't stop the batch evaluation - Individual item failures are tracked but don't stop processing - - Fetch failures are retried with exponential backoff + - Fetch failures are retried up to `max_retries` times - All scores are automatically flushed to Langfuse at the end - - The resume mechanism uses timestamp-based filtering to avoid duplicates + - Resuming uses the pagination cursor, so items are neither skipped nor + evaluated twice """ runner = BatchEvaluationRunner(self) @@ -3401,13 +3391,12 @@ def composite_evaluator(*, item, evaluations): evaluators=evaluators, filter=filter, fetch_batch_size=fetch_batch_size, - fetch_trace_fields=fetch_trace_fields, + fields=fields, max_items=max_items, max_concurrency=max_concurrency, composite_evaluator=composite_evaluator, metadata=metadata, _add_observation_scores_to_trace=_add_observation_scores_to_trace, - _additional_trace_tags=_additional_trace_tags, max_retries=max_retries, verbose=verbose, resume_from=resume_from, diff --git a/langfuse/batch_evaluation.py b/langfuse/batch_evaluation.py index 723b45757..025748de4 100644 --- a/langfuse/batch_evaluation.py +++ b/langfuse/batch_evaluation.py @@ -1,9 +1,10 @@ """Batch evaluation functionality for Langfuse. This module provides comprehensive batch evaluation capabilities for running evaluations -on traces and observations fetched from Langfuse. It includes type definitions, -protocols, result classes, and the implementation for large-scale evaluation workflows -with error handling, retry logic, and resume capability. +on observations fetched from Langfuse via the v2 observations API +(`GET /api/public/v2/observations`). It includes type definitions, protocols, result +classes, and the implementation for large-scale evaluation workflows with error +handling, retry logic, and resume capability. """ import asyncio @@ -15,40 +16,52 @@ Awaitable, Dict, List, + Literal, Optional, Protocol, - Set, Tuple, Union, - cast, + get_args, ) -from langfuse.api import ( - ObservationsView, - TraceWithFullDetails, -) +from langfuse.api import ObservationV2 from langfuse.experiment import Evaluation, EvaluatorFunction from langfuse.logger import langfuse_logger as logger if TYPE_CHECKING: from langfuse._client.client import Langfuse +BatchEvaluationScope = Literal["observations", "root_observations"] +"""Which observations a batch evaluation runs on. + +- ``"observations"``: every observation matching the filter. Scores are attached + to the observation (``observation_id`` + ``trace_id``). +- ``"root_observations"``: only logical root observations, i.e. one per trace. + Scores are attached to the trace (``trace_id`` only), which makes this the + replacement for trace-level batch evaluation. +""" + +DEFAULT_BATCH_EVALUATION_FIELDS = "core,basic,io,metadata" +"""Default v2 observation field groups fetched for batch evaluation. + +Includes ``io`` (raw ``input``/``output`` strings) and ``metadata`` so that mappers +work without extra configuration. Available groups: core, basic, time, io, +metadata, model, usage, prompt, metrics, trace_context. +""" + class EvaluatorInputs: """Input data structure for evaluators, returned by mapper functions. - This class provides a strongly-typed container for transforming API response - objects (traces, observations) into the standardized format expected - by evaluator functions. It ensures consistent access to input, output, expected - output, and metadata regardless of the source entity type. + This class provides a strongly-typed container for transforming `ObservationV2` + objects returned by the v2 observations API into the standardized format + expected by evaluator functions. Attributes: - input: The input data that was provided to generate the output being evaluated. - For traces, this might be the initial prompt or request. For observations, - this could be the span's input. The exact meaning depends on your use case. - output: The actual output that was produced and needs to be evaluated. - For traces, this is typically the final response. For observations, - this might be the generation output or span result. + input: The input data that was provided to generate the output being evaluated, + for example the observation's input. + output: The actual output that was produced and needs to be evaluated, + for example the generation output or span result. expected_output: Optional ground truth or expected result for comparison. Used by evaluators to assess correctness. May be None if no ground truth is available for the entity being evaluated. @@ -57,38 +70,31 @@ class EvaluatorInputs: or any other relevant data that evaluators might use. Examples: - Simple mapper for traces: + Simple mapper for root observations: ```python from langfuse import EvaluatorInputs - def trace_mapper(trace): + def root_mapper(*, item): return EvaluatorInputs( - input=trace.input, - output=trace.output, + input=item.input, # raw string as returned by the API + output=item.output, expected_output=None, # No ground truth available - metadata={"user_id": trace.user_id, "tags": trace.tags} + metadata={"trace_id": item.trace_id, "user_id": item.user_id}, ) ``` - Mapper for observations extracting specific fields: + Mapper that decodes JSON input/output: ```python - def observation_mapper(observation): - # Extract input/output from observation's data - input_data = observation.input if hasattr(observation, 'input') else None - output_data = observation.output if hasattr(observation, 'output') else None + import json + def observation_mapper(*, item): return EvaluatorInputs( - input=input_data, - output=output_data, + input=json.loads(item.input) if item.input else None, + output=json.loads(item.output) if item.output else None, expected_output=None, - metadata={ - "observation_type": observation.type, - "model": observation.model, - "latency_ms": observation.end_time - observation.start_time - } + metadata={"observation_type": item.type, "name": item.name}, ) ``` - ``` Note: All arguments must be passed as keywords when instantiating this class. @@ -122,34 +128,40 @@ def __init__( class MapperFunction(Protocol): """Protocol defining the interface for mapper functions in batch evaluation. - Mapper functions transform API response objects (traces or observations) - into the standardized EvaluatorInputs format that evaluators expect. This abstraction - allows you to define how to extract and structure evaluation data from different - entity types. + Mapper functions transform `ObservationV2` objects from the v2 observations API + into the standardized EvaluatorInputs format that evaluators expect. Mapper functions must: - - Accept a single item parameter (trace, observation) + - Accept a single keyword argument `item` (an `ObservationV2`) - Return an EvaluatorInputs instance with input, output, expected_output, metadata - Can be either synchronous or asynchronous - Should handle missing or malformed data gracefully + + Notes on `ObservationV2`: + - `input` and `output` are raw strings exactly as stored; the SDK does not + parse them. Use `json.loads` in the mapper if you need structured data. + - Only the requested field groups (see the `fields` argument of + `Langfuse.run_batched_evaluation`) are populated; fields of other groups + are None. + - `metadata` values longer than 200 characters are truncated by the API. + - Price fields (`input_price`, `output_price`, `total_price`) are strings. """ def __call__( self, *, - item: Union["TraceWithFullDetails", "ObservationsView"], + item: ObservationV2, **kwargs: Dict[str, Any], ) -> Union[EvaluatorInputs, Awaitable[EvaluatorInputs]]: - """Transform an API response object into evaluator inputs. + """Transform an observation into evaluator inputs. - This method defines how to extract evaluation-relevant data from the raw - API response object. The implementation should map entity-specific fields - to the standardized input/output/expected_output/metadata structure. + This method defines how to extract evaluation-relevant data from the + observation. The implementation should map its fields to the standardized + input/output/expected_output/metadata structure. Args: - item: The API response object to transform. The type depends on the scope: - - TraceWithFullDetails: When evaluating traces - - ObservationsView: When evaluating observations + item: The `ObservationV2` to transform. With + `scope="root_observations"` this is the root observation of a trace. Returns: EvaluatorInputs: A structured container with: @@ -162,48 +174,45 @@ def __call__( (for async mappers that need to fetch additional data). Examples: - Basic trace mapper: + Basic root observation mapper: ```python - def map_trace(trace): + def map_root(*, item): return EvaluatorInputs( - input=trace.input, - output=trace.output, + input=item.input, + output=item.output, expected_output=None, - metadata={"trace_id": trace.id, "user": trace.user_id} + metadata={"trace_id": item.trace_id, "user": item.user_id} ) ``` Observation mapper with conditional logic: ```python - def map_observation(observation): - # Extract fields based on observation type - if observation.type == "GENERATION": - input_data = observation.input - output_data = observation.output + import json + + def map_observation(*, item): + if item.type == "GENERATION": + input_data = json.loads(item.input) if item.input else None else: - # For other types, use different fields - input_data = observation.metadata.get("input") - output_data = observation.metadata.get("output") + input_data = item.input return EvaluatorInputs( input=input_data, - output=output_data, + output=item.output, expected_output=None, - metadata={"obs_id": observation.id, "type": observation.type} + metadata={"obs_id": item.id, "type": item.type} ) ``` Async mapper (if additional processing needed): ```python - async def map_trace_async(trace): - # Could do async processing here if needed - processed_output = await some_async_transformation(trace.output) + async def map_async(*, item): + processed_output = await some_async_transformation(item.output) return EvaluatorInputs( - input=trace.input, + input=item.input, output=processed_output, expected_output=None, - metadata={"trace_id": trace.id} + metadata={"trace_id": item.trace_id} ) ``` """ @@ -452,97 +461,66 @@ def __init__( class BatchEvaluationResumeToken: - """Token for resuming a failed batch evaluation run. + """Token for resuming an interrupted or limited batch evaluation run. + + The v2 observations API returns observations ordered by start time, newest + first, and paginates with an opaque cursor. The token stores the cursor of + the next page that has not been processed yet, so a resumed run continues + exactly where the previous one stopped, without re-evaluating or skipping + items, even if new observations were ingested in the meantime. - This class encapsulates all the information needed to resume a batch evaluation - that was interrupted or failed partway through. It uses timestamp-based filtering - to avoid re-processing items that were already evaluated, even if the underlying - dataset changed between runs. + A token is returned when a run stops because a batch fetch failed after all + retries (`completed=False`) or because `max_items` was reached while more + items exist (`has_more_items=True`). Attributes: - scope: The type of items being evaluated ("traces", "observations"). - filter: The original JSON filter string used to query items. - last_processed_timestamp: ISO 8601 timestamp of the last successfully processed item. - Used to construct a filter that only fetches items after this timestamp. - last_processed_id: The ID of the last successfully processed item, for reference. - items_processed: Count of items successfully processed before interruption. + scope: The scope of the run ("observations" or "root_observations"). + Resuming requires the same scope. + filter: The original JSON filter string used to query items. Pass the + same filter when resuming. + cursor: Cursor of the next page to fetch. None if no page was fetched yet. + last_processed_timestamp: ISO 8601 start time of the oldest processed + observation. Only used to resume when `cursor` is None, by fetching + observations that started strictly before this timestamp. + last_processed_id: The ID of the last processed observation, for reference. + items_processed: Number of items successfully processed so far, including + items processed by the runs this token was resumed from. Examples: - Resuming a failed batch evaluation: + Resuming a run that stopped early: ```python - # Initial run that fails partway through - try: - result = client.run_batched_evaluation( - scope="traces", - mapper=my_mapper, - evaluators=[evaluator1, evaluator2], - filter='{"tags": ["production"]}', - max_items=10000 - ) - except Exception as e: - print(f"Evaluation failed: {e}") - - # Save the resume token - if result.resume_token: - # Store resume token for later (e.g., in a file or database) - import json - with open("resume_token.json", "w") as f: - json.dump({ - "scope": result.resume_token.scope, - "filter": result.resume_token.filter, - "last_timestamp": result.resume_token.last_processed_timestamp, - "last_id": result.resume_token.last_processed_id, - "items_done": result.resume_token.items_processed - }, f) - - # Later, resume from where it left off - with open("resume_token.json") as f: - token_data = json.load(f) - - resume_token = BatchEvaluationResumeToken( - scope=token_data["scope"], - filter=token_data["filter"], - last_processed_timestamp=token_data["last_timestamp"], - last_processed_id=token_data["last_id"], - items_processed=token_data["items_done"] - ) - - # Resume the evaluation result = client.run_batched_evaluation( - scope="traces", + scope="root_observations", mapper=my_mapper, evaluators=[evaluator1, evaluator2], - resume_from=resume_token + filter=my_filter, + max_items=10000, ) - print(f"Processed {result.total_items_processed} additional items") + if result.resume_token: + result = client.run_batched_evaluation( + scope="root_observations", + mapper=my_mapper, + evaluators=[evaluator1, evaluator2], + filter=my_filter, + resume_from=result.resume_token, + ) ``` - Handling partial completion: + Persisting a token between processes: ```python - result = client.run_batched_evaluation(...) + import json - if not result.completed: - print(f"Evaluation incomplete. Processed {result.resume_token.items_processed} items") - print(f"Last item: {result.resume_token.last_processed_id}") - print(f"Resume from: {result.resume_token.last_processed_timestamp}") + token = result.resume_token + with open("resume_token.json", "w") as f: + json.dump(vars(token), f) - # Optionally retry automatically - if result.resume_token: - print("Retrying...") - result = client.run_batched_evaluation( - scope=result.resume_token.scope, - mapper=my_mapper, - evaluators=my_evaluators, - resume_from=result.resume_token - ) + with open("resume_token.json") as f: + token = BatchEvaluationResumeToken(**json.load(f)) ``` Note: All arguments must be passed as keywords when instantiating this class. - The timestamp-based approach means that items created after the initial run - but before the timestamp will be skipped. This is intentional to avoid - duplicates and ensure consistent evaluation. """ def __init__( @@ -553,21 +531,24 @@ def __init__( last_processed_timestamp: str, last_processed_id: str, items_processed: int, + cursor: Optional[str] = None, ): """Initialize BatchEvaluationResumeToken with the provided state. Args: - scope: The scope type ("traces", "observations"). + scope: The scope of the run ("observations" or "root_observations"). filter: The original JSON filter string. - last_processed_timestamp: ISO 8601 timestamp of last processed item. + last_processed_timestamp: ISO 8601 start time of the oldest processed item. last_processed_id: ID of last processed item. items_processed: Count of items processed before interruption. + cursor: Cursor of the next page to fetch. Note: All arguments must be provided as keywords. """ self.scope = scope self.filter = filter + self.cursor = cursor self.last_processed_timestamp = last_processed_timestamp self.last_processed_id = last_processed_id self.items_processed = items_processed @@ -588,13 +569,16 @@ class BatchEvaluationResult: total_composite_scores_created: Scores created by the composite evaluator. total_evaluations_failed: Number of individual evaluator failures across all items. evaluator_stats: List of per-evaluator statistics (success/failure rates, scores created). - resume_token: Token for resuming if evaluation was interrupted (None if completed). - completed: True if all items were processed, False if stopped early or failed. + resume_token: Token for continuing the run. Set when a batch fetch failed + (`completed=False`) or when `max_items` was reached while more items + exist (`has_more_items=True`); None otherwise. + completed: False if the run stopped because a batch fetch failed, True otherwise + (including when it stopped at `max_items`). duration_seconds: Total time taken to execute the batch evaluation. - failed_item_ids: List of IDs for items that failed evaluation. + failed_item_ids: List of observation IDs for items that failed evaluation. error_summary: Dictionary mapping error types to occurrence counts. has_more_items: True if max_items limit was reached but more items exist. - item_evaluations: Dictionary mapping item IDs to their evaluation results (both regular and composite). + item_evaluations: Dictionary mapping observation IDs to their evaluation results (both regular and composite). Examples: Basic result inspection: @@ -644,9 +628,7 @@ class BatchEvaluationResult: if result.resume_token: print(f"Processed {result.resume_token.items_processed} items before failure") - print(f"Use resume_from parameter to continue from:") - print(f" Timestamp: {result.resume_token.last_processed_timestamp}") - print(f" Last ID: {result.resume_token.last_processed_id}") + print("Pass it as resume_from to continue") if result.has_more_items: print(f"ℹ️ More items available beyond max_items limit") @@ -839,57 +821,70 @@ def __init__(self, client: "Langfuse"): async def run_async( self, *, - scope: str, + scope: BatchEvaluationScope, mapper: MapperFunction, evaluators: List[EvaluatorFunction], filter: Optional[str] = None, fetch_batch_size: int = 50, - fetch_trace_fields: Optional[str] = "io", + fields: Optional[str] = DEFAULT_BATCH_EVALUATION_FIELDS, max_items: Optional[int] = None, max_concurrency: int = 5, composite_evaluator: Optional[CompositeEvaluatorFunction] = None, metadata: Optional[Dict[str, Any]] = None, _add_observation_scores_to_trace: bool = False, - _additional_trace_tags: Optional[List[str]] = None, max_retries: int = 3, verbose: bool = False, resume_from: Optional[BatchEvaluationResumeToken] = None, ) -> BatchEvaluationResult: - """Run batch evaluation asynchronously using legacy read APIs. + """Run batch evaluation asynchronously. This is the main implementation method that orchestrates the entire batch - evaluation process: fetching items, mapping, evaluating, creating scores, - and tracking statistics. - - This runner reads traces from `GET /api/public/traces` and observations - from the legacy `GET /api/public/observations` endpoint. It is supported - with Langfuse platform v3 and is not yet supported with platform v4. + evaluation process: fetching items from `GET /api/public/v2/observations` + with cursor pagination, mapping, evaluating, creating scores, and tracking + statistics. Args: - scope: The type of items to evaluate ("traces", "observations"). - mapper: Function to transform API response items to evaluator inputs. + scope: Which observations to evaluate ("observations", "root_observations"). + mapper: Function to transform `ObservationV2` items to evaluator inputs. evaluators: List of evaluation functions to run on each item. - filter: JSON filter string for querying items. - fetch_batch_size: Number of items to fetch per API call. - fetch_trace_fields: Comma-separated list of fields to include when fetching traces. Available field groups: 'core' (always included), 'io' (input, output, metadata), 'scores', 'observations', 'metrics'. If not specified, all fields are returned. Example: 'core,scores,metrics'. Note: Excluded 'observations' or 'scores' fields return empty arrays; excluded 'metrics' returns -1 for 'totalCost' and 'latency'. Only relevant if scope is 'traces'. Default: 'io' + filter: JSON filter string (v2 observations filter schema). + fetch_batch_size: Number of items to fetch per API call (max 1000). + fields: Comma-separated v2 observation field groups to fetch. max_items: Maximum number of items to process (None = all). max_concurrency: Maximum number of concurrent evaluations. composite_evaluator: Optional function to create composite scores. metadata: Metadata to add to all created scores. _add_observation_scores_to_trace: Private option to duplicate - observation-level scores onto the parent trace. - _additional_trace_tags: Private option to add tags on traces via - ingestion trace-create events. + observation-level scores onto the parent trace. Only applies to + scope="observations". max_retries: Maximum retries for failed batch fetches. verbose: If True, log progress to console. - resume_from: Resume token from a previous failed run. + resume_from: Resume token from a previous run. Returns: BatchEvaluationResult with comprehensive statistics. + + Raises: + ValueError: If the scope is invalid, the filter is not a JSON array, or + the resume token was created for a different scope. """ start_time = time.time() - # Initialize tracking variables + if scope not in get_args(BatchEvaluationScope): + raise ValueError( + f"Invalid scope: {scope!r}. Expected one of " + f"{', '.join(repr(s) for s in get_args(BatchEvaluationScope))}." + ) + if resume_from is not None and resume_from.scope != scope: + raise ValueError( + f"Resume token was created for scope {resume_from.scope!r}, " + f"cannot resume with scope {scope!r}." + ) + + effective_filter = self._build_filter( + filter=filter, scope=scope, resume_from=resume_from + ) + total_items_fetched = 0 total_items_processed = 0 total_items_failed = 0 @@ -899,8 +894,8 @@ async def run_async( failed_item_ids: List[str] = [] error_summary: Dict[str, int] = {} item_evaluations: Dict[str, List[Evaluation]] = {} + previously_processed = resume_from.items_processed if resume_from else 0 - # Initialize evaluator stats evaluator_stats_dict = { getattr(evaluator, "__name__", "unknown_evaluator"): EvaluatorStats( name=getattr(evaluator, "__name__", "unknown_evaluator") @@ -908,65 +903,60 @@ async def run_async( for evaluator in evaluators } - # Handle resume token by modifying filter - effective_filter = self._build_timestamp_filter(filter, resume_from) - normalized_additional_trace_tags = ( - self._dedupe_tags(_additional_trace_tags) - if _additional_trace_tags is not None - else [] - ) - updated_trace_ids: Set[str] = set() - - # Create semaphore for concurrency control semaphore = asyncio.Semaphore(max_concurrency) - # Pagination state - page = 1 + cursor: Optional[str] = resume_from.cursor if resume_from else None has_more = True - last_item_timestamp: Optional[str] = None - last_item_id: Optional[str] = None + last_item_timestamp = ( + resume_from.last_processed_timestamp if resume_from else "" + ) + last_item_id = resume_from.last_processed_id if resume_from else "" + batch_number = 0 if verbose: logger.info("Starting batch evaluation on %s", scope) - if scope == "traces" and fetch_trace_fields: - logger.info("Fetching trace fields: %s", fetch_trace_fields) + if fields: + logger.info("Fetching observation fields: %s", fields) if resume_from: logger.info( - "Resuming from %s (%s items already processed)", - resume_from.last_processed_timestamp, + "Resuming after %s (%s items already processed)", + resume_from.last_processed_timestamp or "start", resume_from.items_processed, ) - # Main pagination loop - while has_more: - # Check if we've reached max_items + def build_resume_token() -> BatchEvaluationResumeToken: + return BatchEvaluationResumeToken( + scope=scope, + filter=filter, + cursor=cursor, + last_processed_timestamp=last_item_timestamp, + last_processed_id=last_item_id, + items_processed=previously_processed + total_items_processed, + ) + + while True: if max_items is not None and total_items_fetched >= max_items: if verbose: logger.info("Reached max_items limit (%s)", max_items) - has_more = True # More items may exist break - # Fetch next batch with retry logic + # Clamping the page size keeps the cursor aligned with the last + # processed item, so resuming after max_items skips nothing. + limit = fetch_batch_size + if max_items is not None: + limit = min(limit, max_items - total_items_fetched) + try: - items = await self._fetch_batch_with_retry( - scope=scope, + items, next_cursor = await self._fetch_batch_with_retry( filter=effective_filter, - page=page, - limit=fetch_batch_size, + cursor=cursor, + limit=limit, max_retries=max_retries, - fields=fetch_trace_fields, + fields=fields, ) except Exception as e: - # Failed after max_retries - create resume token and return - error_msg = f"Failed to fetch batch after {max_retries} retries" - logger.error("%s: %s", error_msg, e) - - resume_token = BatchEvaluationResumeToken( - scope=scope, - filter=filter, # Original filter, not modified - last_processed_timestamp=last_item_timestamp or "", - last_processed_id=last_item_id or "", - items_processed=total_items_processed, + logger.error( + "Failed to fetch batch after %s retries: %s", max_retries, e ) return self._build_result( @@ -977,47 +967,25 @@ async def run_async( total_composite_scores_created=total_composite_scores_created, total_evaluations_failed=total_evaluations_failed, evaluator_stats_dict=evaluator_stats_dict, - resume_token=resume_token, + resume_token=build_resume_token(), completed=False, start_time=start_time, failed_item_ids=failed_item_ids, error_summary=error_summary, - has_more_items=has_more, + has_more_items=False, item_evaluations=item_evaluations, ) - # Check if we got any items - if not items: - has_more = False - if verbose: - logger.info("No more items to fetch") - break - + batch_number += 1 total_items_fetched += len(items) if verbose: - logger.info("Fetched batch %s (%s items)", page, len(items)) - - # Limit items if max_items would be exceeded - items_to_process = items - if max_items is not None: - remaining_capacity = max_items - total_items_processed - if len(items) > remaining_capacity: - items_to_process = items[:remaining_capacity] - if verbose: - logger.info( - "Limiting batch to %s items to respect max_items=%s", - len(items_to_process), - max_items, - ) + logger.info("Fetched batch %s (%s items)", batch_number, len(items)) - # Process items concurrently async def process_item( - item: Union[TraceWithFullDetails, ObservationsView], + item: ObservationV2, ) -> Tuple[str, Union[Tuple[int, int, int, List[Evaluation]], Exception]]: - """Process a single item and return (item_id, result).""" async with semaphore: - item_id = self._get_item_id(item, scope) try: result = await self._process_batch_evaluation_item( item=item, @@ -1029,25 +997,20 @@ async def process_item( _add_observation_scores_to_trace=_add_observation_scores_to_trace, evaluator_stats_dict=evaluator_stats_dict, ) - return (item_id, result) + return (item.id, result) except Exception as e: - return (item_id, e) + return (item.id, e) - # Run all items in batch concurrently - tasks = [process_item(item) for item in items_to_process] - results = await asyncio.gather(*tasks) + results = await asyncio.gather(*[process_item(item) for item in items]) - # Process results and update statistics - for item, (item_id, result) in zip(items_to_process, results): + for item, (item_id, result) in zip(items, results): if isinstance(result, Exception): - # Item processing failed total_items_failed += 1 failed_item_ids.append(item_id) error_type = type(result).__name__ error_summary[error_type] = error_summary.get(error_type, 0) + 1 logger.warning("Item %s failed: %s", item_id, result) else: - # Item processed successfully total_items_processed += 1 scores_created, composite_created, evals_failed, evaluations = ( result @@ -1055,36 +1018,19 @@ async def process_item( total_scores_created += scores_created total_composite_scores_created += composite_created total_evaluations_failed += evals_failed - - # Store evaluations for this item item_evaluations[item_id] = evaluations - if normalized_additional_trace_tags: - trace_id = ( - item_id - if scope == "traces" - else cast(ObservationsView, item).trace_id - ) - - if trace_id and trace_id not in updated_trace_ids: - self.client._create_trace_tags_via_ingestion( - trace_id=trace_id, - tags=normalized_additional_trace_tags, - ) - updated_trace_ids.add(trace_id) - - # Update last processed tracking - last_item_timestamp = self._get_item_timestamp(item, scope) - last_item_id = item_id + if items: + last_item_timestamp = items[-1].start_time.isoformat() + last_item_id = items[-1].id if verbose: if max_items is not None and max_items > 0: - progress_pct = total_items_processed / max_items * 100 logger.info( "Progress: %s/%s items (%.1f%%), %s scores created", total_items_processed, max_items, - progress_pct, + total_items_processed / max_items * 100, total_scores_created, ) else: @@ -1094,40 +1040,22 @@ async def process_item( total_scores_created, ) - # Check if we should continue to next page - if len(items) < fetch_batch_size: - # Last page - no more items available + cursor = next_cursor + if cursor is None or not items: has_more = False - else: - page += 1 - - # Check max_items again before next fetch - if max_items is not None and total_items_fetched >= max_items: - has_more = True # More items exist but we're stopping - break + break - # Flush all scores to Langfuse if verbose: logger.info("Flushing scores to Langfuse...") self.client.flush() - # Build final result - duration = time.time() - start_time - if verbose: logger.info( "Batch evaluation complete: %s items processed in %.2fs", total_items_processed, - duration, + time.time() - start_time, ) - # Completed successfully if we either: - # 1. Ran out of items (has_more is False), OR - # 2. Hit max_items limit (intentionally stopped) - completed_successfully = not has_more or ( - max_items is not None and total_items_fetched >= max_items - ) - return self._build_result( total_items_fetched=total_items_fetched, total_items_processed=total_items_processed, @@ -1136,69 +1064,55 @@ async def process_item( total_composite_scores_created=total_composite_scores_created, total_evaluations_failed=total_evaluations_failed, evaluator_stats_dict=evaluator_stats_dict, - resume_token=None, # No resume needed on successful completion - completed=completed_successfully, + resume_token=build_resume_token() if has_more else None, + completed=True, start_time=start_time, failed_item_ids=failed_item_ids, error_summary=error_summary, - has_more_items=( - has_more and max_items is not None and total_items_fetched >= max_items - ), + has_more_items=has_more, item_evaluations=item_evaluations, ) async def _fetch_batch_with_retry( self, *, - scope: str, filter: Optional[str], - page: int, + cursor: Optional[str], limit: int, max_retries: int, fields: Optional[str], - ) -> List[Union[TraceWithFullDetails, ObservationsView]]: - """Fetch a batch of items with retry logic. + ) -> Tuple[List[ObservationV2], Optional[str]]: + """Fetch one page of observations from the v2 observations API. Args: - scope: The type of items ("traces", "observations"). filter: JSON filter string for querying. - page: Page number (1-indexed). - limit: Number of items per page. + cursor: Cursor returned with the previous page; None for the first page. + limit: Number of items to request. max_retries: Maximum number of retry attempts. - verbose: Whether to log retry attempts. - fields: Trace fields to fetch + fields: Comma-separated v2 field groups to include. Returns: - List of items from the API. + Tuple of the page's observations and the cursor for the next page + (None when there are no more pages). Raises: Exception: If all retry attempts fail. """ - if scope == "traces": - response = self.client.api.trace.list( - page=page, - limit=limit, - filter=filter, - request_options={"max_retries": max_retries}, - fields=fields, - ) # type: ignore - return list(response.data) # type: ignore - elif scope == "observations": - response = self.client.api.legacy.observations_v1.get_many( - page=page, - limit=limit, - filter=filter, - request_options={"max_retries": max_retries}, - ) # type: ignore - return list(response.data) # type: ignore - else: - error_message = f"Invalid scope: {scope}" - raise ValueError(error_message) + response = await asyncio.to_thread( + self.client.api.observations.get_many, + fields=fields, + limit=limit, + cursor=cursor, + filter=filter, + request_options={"max_retries": max_retries}, + ) + + return list(response.data), response.meta.cursor async def _process_batch_evaluation_item( self, - item: Union[TraceWithFullDetails, ObservationsView], - scope: str, + item: ObservationV2, + scope: BatchEvaluationScope, mapper: MapperFunction, evaluators: List[EvaluatorFunction], composite_evaluator: Optional[CompositeEvaluatorFunction], @@ -1209,8 +1123,8 @@ async def _process_batch_evaluation_item( """Process a single item: map, evaluate, create scores. Args: - item: The API response object to evaluate. - scope: The type of item ("traces", "observations"). + item: The observation to evaluate. + scope: The scope of the run. mapper: Function to transform item to evaluator inputs. evaluators: List of evaluator functions. composite_evaluator: Optional composite evaluator function. @@ -1225,14 +1139,15 @@ async def _process_batch_evaluation_item( Raises: Exception: If mapping fails or item processing encounters fatal error. """ + if not item.trace_id: + raise ValueError(f"Observation {item.id} has no trace_id") + scores_created = 0 composite_scores_created = 0 evaluations_failed = 0 - # Run mapper to transform item evaluator_inputs = await self._run_mapper(mapper, item) - # Run all evaluators evaluations: List[Evaluation] = [] for evaluator in evaluators: evaluator_name = getattr(evaluator, "__name__", "unknown_evaluator") @@ -1253,31 +1168,24 @@ async def _process_batch_evaluation_item( evaluations.extend(eval_results) except Exception as e: - # Evaluator failed - log warning and continue with other evaluators stats.failed_runs += 1 evaluations_failed += 1 logger.warning( "Evaluator %s failed on item %s: %s", evaluator_name, - self._get_item_id(item, scope), + item.id, e, ) - # Create scores for item-level evaluations - item_id = self._get_item_id(item, scope) for evaluation in evaluations: scores_created += self._create_score_for_scope( scope=scope, - item_id=item_id, - trace_id=cast(ObservationsView, item).trace_id - if scope == "observations" - else None, + item=item, evaluation=evaluation, additional_metadata=metadata, add_observation_score_to_trace=_add_observation_scores_to_trace, ) - # Run composite evaluator if provided and we have evaluations if composite_evaluator and evaluations: try: composite_evals = await self._run_composite_evaluator( @@ -1289,24 +1197,19 @@ async def _process_batch_evaluation_item( evaluations=evaluations, ) - # Create scores for all composite evaluations for composite_eval in composite_evals: composite_scores_created += self._create_score_for_scope( scope=scope, - item_id=item_id, - trace_id=cast(ObservationsView, item).trace_id - if scope == "observations" - else None, + item=item, evaluation=composite_eval, additional_metadata=metadata, add_observation_score_to_trace=_add_observation_scores_to_trace, ) - # Add composite evaluations to the list evaluations.extend(composite_evals) except Exception as e: - logger.warning("Composite evaluator failed on item %s: %s", item_id, e) + logger.warning("Composite evaluator failed on item %s: %s", item.id, e) return ( scores_created, @@ -1337,11 +1240,9 @@ async def _run_evaluator_internal( """ result = evaluator(**kwargs) - # Handle async evaluators if asyncio.iscoroutine(result): result = await result - # Normalize to list if isinstance(result, (dict, Evaluation)): return [result] # type: ignore elif isinstance(result, list): @@ -1352,13 +1253,13 @@ async def _run_evaluator_internal( async def _run_mapper( self, mapper: MapperFunction, - item: Union[TraceWithFullDetails, ObservationsView], + item: ObservationV2, ) -> EvaluatorInputs: """Run mapper function (handles both sync and async mappers). Args: mapper: The mapper function to run. - item: The API response object to map. + item: The observation to map. Returns: EvaluatorInputs instance. @@ -1406,7 +1307,6 @@ async def _run_composite_evaluator( if asyncio.iscoroutine(result): result = await result - # Normalize to list (same as regular evaluator) if isinstance(result, (dict, Evaluation)): return [result] # type: ignore elif isinstance(result, list): @@ -1417,19 +1317,20 @@ async def _run_composite_evaluator( def _create_score_for_scope( self, *, - scope: str, - item_id: str, - trace_id: Optional[str] = None, + scope: BatchEvaluationScope, + item: ObservationV2, evaluation: Evaluation, additional_metadata: Optional[Dict[str, Any]], add_observation_score_to_trace: bool = False, ) -> int: - """Create a score linked to the appropriate entity based on scope. + """Create a score linked to the entity that the scope evaluates. + + `root_observations` scores the trace; `observations` scores the + observation and optionally duplicates the score onto its trace. Args: - scope: The type of entity ("traces", "observations"). - item_id: The ID of the entity. - trace_id: The trace ID of the entity; required if scope=observations + scope: The scope of the run. + item: The evaluated observation. evaluation: The evaluation result to create a score from. additional_metadata: Additional metadata to merge with evaluation metadata. add_observation_score_to_trace: Whether to duplicate observation @@ -1438,165 +1339,96 @@ def _create_score_for_scope( Returns: Number of score events created. """ - # Merge metadata score_metadata = { **(evaluation.metadata or {}), **(additional_metadata or {}), } + score_kwargs: Dict[str, Any] = { + "name": evaluation.name, + "value": evaluation.value, + "comment": evaluation.comment, + "metadata": score_metadata, + "data_type": evaluation.data_type, + "config_id": evaluation.config_id, + } - if scope == "traces": - self.client.create_score( - trace_id=item_id, - name=evaluation.name, - value=evaluation.value, # type: ignore - comment=evaluation.comment, - metadata=score_metadata, - data_type=evaluation.data_type, # type: ignore[arg-type] - config_id=evaluation.config_id, - ) + if scope == "root_observations": + self.client.create_score(trace_id=item.trace_id, **score_kwargs) return 1 - elif scope == "observations": - self.client.create_score( - observation_id=item_id, - trace_id=trace_id, - name=evaluation.name, - value=evaluation.value, # type: ignore - comment=evaluation.comment, - metadata=score_metadata, - data_type=evaluation.data_type, # type: ignore[arg-type] - config_id=evaluation.config_id, - ) - score_count = 1 - - if add_observation_score_to_trace and trace_id: - self.client.create_score( - trace_id=trace_id, - name=evaluation.name, - value=evaluation.value, # type: ignore - comment=evaluation.comment, - metadata=score_metadata, - data_type=evaluation.data_type, # type: ignore[arg-type] - config_id=evaluation.config_id, - ) - score_count += 1 - return score_count + self.client.create_score( + observation_id=item.id, trace_id=item.trace_id, **score_kwargs + ) + if not add_observation_score_to_trace: + return 1 - return 0 + self.client.create_score(trace_id=item.trace_id, **score_kwargs) + return 2 - def _build_timestamp_filter( - self, - original_filter: Optional[str], + @staticmethod + def _build_filter( + *, + filter: Optional[str], + scope: BatchEvaluationScope, resume_from: Optional[BatchEvaluationResumeToken], ) -> Optional[str]: - """Build filter with timestamp constraint for resume capability. + """Combine the user filter with the scope and resume constraints. - Args: - original_filter: The original JSON filter string. - resume_from: Optional resume token with timestamp information. - - Returns: - Modified filter string with timestamp constraint, or original filter. - """ - if not resume_from: - return original_filter - - # Parse original filter (should be array) or create empty array - try: - filter_list = json.loads(original_filter) if original_filter else [] - if not isinstance(filter_list, list): - logger.warning( - "Filter should be a JSON array, got: %s", type(filter_list).__name__ - ) - filter_list = [] - except json.JSONDecodeError: - logger.warning( - "Invalid JSON in original filter, ignoring: %s", original_filter - ) - filter_list = [] - - # Add timestamp constraint to filter array - timestamp_field = self._get_timestamp_field_for_scope(resume_from.scope) - timestamp_filter = { - "type": "datetime", - "column": timestamp_field, - "operator": ">", - "value": resume_from.last_processed_timestamp, - } - filter_list.append(timestamp_filter) - - return json.dumps(filter_list) - - @staticmethod - def _get_item_id( - item: Union[TraceWithFullDetails, ObservationsView], - scope: str, - ) -> str: - """Extract ID from item based on scope. + Constraints are added as JSON filter conditions rather than query + parameters because the API drops a query-parameter filter whenever the + JSON filter has a condition on the same column. Args: - item: The API response object. - scope: The type of item. + filter: The user-provided JSON filter string (a JSON array). + scope: The scope of the run. + resume_from: Optional resume token. Returns: - The item's ID. - """ - return item.id - - @staticmethod - def _get_item_timestamp( - item: Union[TraceWithFullDetails, ObservationsView], - scope: str, - ) -> str: - """Extract timestamp from item based on scope. - - Args: - item: The API response object. - scope: The type of item. - - Returns: - ISO 8601 timestamp string. - """ - if scope == "traces": - # Type narrowing for traces - if hasattr(item, "timestamp"): - return item.timestamp.isoformat() # type: ignore[attr-defined] - elif scope == "observations": - # Type narrowing for observations - if hasattr(item, "start_time"): - return item.start_time.isoformat() # type: ignore[attr-defined] - return "" - - @staticmethod - def _get_timestamp_field_for_scope(scope: str) -> str: - """Get the timestamp field name for filtering based on scope. - - Args: - scope: The type of items. + The JSON filter string to send, or None if there are no conditions. - Returns: - The field name to use in filters. + Raises: + ValueError: If the filter is not a JSON array. """ - if scope == "traces": - return "timestamp" - elif scope == "observations": - return "start_time" - return "timestamp" # Default - - @staticmethod - def _dedupe_tags(tags: Optional[List[str]]) -> List[str]: - """Deduplicate tags while preserving order.""" - if tags is None: - return [] + conditions: List[Any] = [] + if filter: + try: + parsed = json.loads(filter) + except json.JSONDecodeError as e: + raise ValueError(f"filter must be a JSON array: {e}") from e + if not isinstance(parsed, list): + raise ValueError( + f"filter must be a JSON array of conditions, got {type(parsed).__name__}" + ) + conditions.extend(parsed) + + if scope == "root_observations": + conditions.append( + { + "type": "boolean", + "column": "isRootObservation", + "operator": "=", + "value": True, + } + ) - deduped: List[str] = [] - seen = set() - for tag in tags: - if tag not in seen: - deduped.append(tag) - seen.add(tag) + # Results are ordered by start time descending, so items that remain + # after the last processed one started before it. The cursor resumes + # exactly; the timestamp is the fallback for tokens without a cursor. + if ( + resume_from is not None + and resume_from.cursor is None + and resume_from.last_processed_timestamp + ): + conditions.append( + { + "type": "datetime", + "column": "startTime", + "operator": "<", + "value": resume_from.last_processed_timestamp, + } + ) - return deduped + return json.dumps(conditions) if conditions else None def _build_result( self, @@ -1625,12 +1457,12 @@ def _build_result( total_composite_scores_created: Scores from composite evaluator. total_evaluations_failed: Individual evaluator failures. evaluator_stats_dict: Per-evaluator statistics. - resume_token: Resume token if incomplete. - completed: Whether evaluation completed fully. + resume_token: Resume token if the run can be continued. + completed: Whether evaluation completed without a fetch failure. start_time: Start time (unix timestamp). failed_item_ids: IDs of failed items. error_summary: Error type counts. - has_more_items: Whether more items exist. + has_more_items: Whether more items exist beyond max_items. item_evaluations: Dictionary mapping item IDs to their evaluation results. Returns: diff --git a/tests/unit/test_batch_evaluation.py b/tests/unit/test_batch_evaluation.py new file mode 100644 index 000000000..2d924de95 --- /dev/null +++ b/tests/unit/test_batch_evaluation.py @@ -0,0 +1,376 @@ +"""Unit tests for batch evaluation on the v2 observations API.""" + +import inspect +import json +from datetime import datetime, timedelta, timezone +from types import SimpleNamespace +from typing import Any, Dict, List, Optional +from unittest.mock import MagicMock + +import pytest + +from langfuse import Langfuse +from langfuse.api import ObservationsV2Response, ObservationV2 +from langfuse.batch_evaluation import ( + DEFAULT_BATCH_EVALUATION_FIELDS, + BatchEvaluationResumeToken, + BatchEvaluationRunner, + EvaluatorInputs, +) +from langfuse.experiment import Evaluation + +BASE_TIME = datetime(2026, 1, 1, tzinfo=timezone.utc) + + +def make_observation( + index: int, + *, + trace_id: Optional[str] = "default", + is_root: bool = False, + start_time: Optional[datetime] = None, +) -> ObservationV2: + return ObservationV2( + id=f"obs-{index}", + trace_id=f"trace-{index}" if trace_id == "default" else trace_id, + start_time=start_time or BASE_TIME + timedelta(seconds=index), + end_time=None, + project_id="project", + parent_observation_id=None if is_root else f"parent-{index}", + type="SPAN", + is_root_observation=is_root, + input=json.dumps({"question": index}), + output=f"answer {index}", + ) + + +class FakeObservationsApi: + """Serves observations newest first with an index cursor, like the v2 API.""" + + def __init__(self, observations: List[ObservationV2]): + self.observations = sorted( + observations, key=lambda o: (o.start_time, o.id), reverse=True + ) + self.calls: List[Dict[str, Any]] = [] + self.fail_on_call: Optional[int] = None + + def get_many(self, **kwargs: Any) -> ObservationsV2Response: + self.calls.append(kwargs) + if self.fail_on_call is not None and len(self.calls) == self.fail_on_call: + raise ConnectionError("fetch failed") + + matching = [o for o in self.observations if self._matches(o, kwargs["filter"])] + start = int(kwargs["cursor"]) if kwargs["cursor"] else 0 + page = matching[start : start + kwargs["limit"]] + has_more = start + kwargs["limit"] < len(matching) + meta = {"cursor": str(start + kwargs["limit"])} if has_more else {} + return ObservationsV2Response(data=page, meta=meta) + + @staticmethod + def _matches(observation: ObservationV2, filter_json: Optional[str]) -> bool: + for condition in json.loads(filter_json) if filter_json else []: + if condition["column"] == "isRootObservation": + if observation.is_root_observation is not condition["value"]: + return False + elif condition["column"] == "startTime": + assert condition["operator"] == "<" + if observation.start_time >= datetime.fromisoformat(condition["value"]): + return False + return True + + +def make_runner(observations: List[ObservationV2]): + api = FakeObservationsApi(observations) + client = SimpleNamespace( + api=SimpleNamespace(observations=api), + create_score=MagicMock(), + flush=MagicMock(), + ) + return BatchEvaluationRunner(client), api, client # type: ignore[arg-type] + + +def mapper(*, item: ObservationV2) -> EvaluatorInputs: + return EvaluatorInputs(input=item.input, output=item.output, metadata={}) + + +def length_evaluator(*, input, output, **kwargs): + return Evaluation(name="length", value=len(output)) + + +async def run(runner: BatchEvaluationRunner, **kwargs: Any): + kwargs.setdefault("scope", "observations") + kwargs.setdefault("mapper", mapper) + kwargs.setdefault("evaluators", [length_evaluator]) + return await runner.run_async(**kwargs) + + +@pytest.mark.asyncio +async def test_observations_scope_paginates_with_cursor_and_scores_observations(): + runner, api, client = make_runner([make_observation(i) for i in range(5)]) + + result = await run(runner, fetch_batch_size=2) + + assert [call["cursor"] for call in api.calls] == [None, "2", "4"] + assert all("page" not in call for call in api.calls) + assert all( + call["fields"] == DEFAULT_BATCH_EVALUATION_FIELDS and call["filter"] is None + for call in api.calls + ) + assert result.completed is True + assert result.has_more_items is False + assert result.resume_token is None + assert result.total_items_fetched == 5 + assert result.total_items_processed == 5 + assert result.total_scores_created == 5 + assert set(result.item_evaluations) == {f"obs-{i}" for i in range(5)} + client.create_score.assert_any_call( + observation_id="obs-3", + trace_id="trace-3", + name="length", + value=len("answer 3"), + comment=None, + metadata={}, + data_type=None, + config_id=None, + ) + client.flush.assert_called_once() + + +@pytest.mark.asyncio +async def test_root_observations_scope_filters_roots_and_scores_traces(): + observations = [make_observation(i, is_root=i % 2 == 0) for i in range(6)] + runner, api, client = make_runner(observations) + user_filter = [ + {"type": "string", "column": "traceName", "operator": "=", "value": "chat"} + ] + + result = await run( + runner, + scope="root_observations", + filter=json.dumps(user_filter), + metadata={"run": "nightly"}, + ) + + assert json.loads(api.calls[0]["filter"]) == user_filter + [ + { + "type": "boolean", + "column": "isRootObservation", + "operator": "=", + "value": True, + } + ] + assert set(result.item_evaluations) == {"obs-0", "obs-2", "obs-4"} + scored = [call.kwargs for call in client.create_score.call_args_list] + assert sorted(kwargs["trace_id"] for kwargs in scored) == [ + "trace-0", + "trace-2", + "trace-4", + ] + assert all("observation_id" not in kwargs for kwargs in scored) + assert all(kwargs["metadata"] == {"run": "nightly"} for kwargs in scored) + + +@pytest.mark.asyncio +async def test_mapper_receives_raw_string_io_and_requested_fields(): + runner, api, _ = make_runner([make_observation(1)]) + seen: List[ObservationV2] = [] + + def recording_mapper(*, item): + seen.append(item) + return mapper(item=item) + + await run(runner, mapper=recording_mapper, fields="core,io") + + assert api.calls[0]["fields"] == "core,io" + assert isinstance(seen[0], ObservationV2) + assert seen[0].input == '{"question": 1}' + + +def test_client_and_runner_share_default_fields(): + client_default = ( + inspect.signature(Langfuse.run_batched_evaluation).parameters["fields"].default + ) + runner_default = ( + inspect.signature(BatchEvaluationRunner.run_async).parameters["fields"].default + ) + + assert client_default == runner_default == DEFAULT_BATCH_EVALUATION_FIELDS + assert "io" in DEFAULT_BATCH_EVALUATION_FIELDS.split(",") + + +@pytest.mark.asyncio +@pytest.mark.parametrize( + ("kwargs", "message"), + [ + ({"scope": "traces"}, "Invalid scope"), + ({"filter": "not json"}, "JSON array"), + ({"filter": '{"tags": ["a"]}'}, "JSON array"), + ( + { + "resume_from": BatchEvaluationResumeToken( + scope="root_observations", + filter=None, + last_processed_timestamp="", + last_processed_id="", + items_processed=0, + ) + }, + "scope", + ), + ], +) +async def test_invalid_arguments_raise_before_fetching(kwargs, message): + runner, api, _ = make_runner([make_observation(1)]) + + with pytest.raises(ValueError, match=message): + await run(runner, **kwargs) + + assert api.calls == [] + + +@pytest.mark.asyncio +async def test_max_items_aligns_pages_and_resume_continues_without_gaps(): + # Shared start times exercise ties that a timestamp-only resume would skip. + observations = [ + make_observation(i, start_time=BASE_TIME + timedelta(seconds=i // 3)) + for i in range(10) + ] + runner, api, _ = make_runner(observations) + + first = await run(runner, max_items=5, fetch_batch_size=3) + + assert [call["limit"] for call in api.calls] == [3, 2] + assert first.completed is True + assert first.has_more_items is True + assert first.total_items_fetched == 5 + token = first.resume_token + assert token is not None + assert token.cursor == "5" + assert token.items_processed == 5 + assert token.scope == "observations" + + second = await run(runner, resume_from=token, fetch_batch_size=3) + + assert second.resume_token is None + assert second.has_more_items is False + assert set(first.item_evaluations).isdisjoint(second.item_evaluations) + assert set(first.item_evaluations) | set(second.item_evaluations) == { + f"obs-{i}" for i in range(10) + } + + +@pytest.mark.asyncio +async def test_max_items_reached_on_last_page_has_no_resume_token(): + runner, _, _ = make_runner([make_observation(i) for i in range(4)]) + + result = await run(runner, max_items=4, fetch_batch_size=2) + + assert result.has_more_items is False + assert result.resume_token is None + + +@pytest.mark.asyncio +async def test_fetch_failure_returns_resume_token_for_failed_page(): + runner, api, _ = make_runner([make_observation(i) for i in range(6)]) + api.fail_on_call = 2 + + failed = await run(runner, fetch_batch_size=2, max_retries=0) + + assert failed.completed is False + assert failed.total_items_processed == 2 + token = failed.resume_token + assert token is not None + assert token.cursor == "2" + assert token.last_processed_id == "obs-4" + + api.fail_on_call = None + resumed = await run(runner, fetch_batch_size=2, resume_from=token) + + assert resumed.completed is True + assert set(resumed.item_evaluations) == {"obs-3", "obs-2", "obs-1", "obs-0"} + assert api.calls[-1]["request_options"] == {"max_retries": 3} + + +@pytest.mark.asyncio +async def test_resume_without_cursor_falls_back_to_start_time_bound(): + runner, api, _ = make_runner([make_observation(i) for i in range(5)]) + token = BatchEvaluationResumeToken( + scope="observations", + filter=None, + last_processed_timestamp=(BASE_TIME + timedelta(seconds=3)).isoformat(), + last_processed_id="obs-3", + items_processed=2, + ) + + result = await run(runner, resume_from=token) + + assert json.loads(api.calls[0]["filter"]) == [ + { + "type": "datetime", + "column": "startTime", + "operator": "<", + "value": token.last_processed_timestamp, + } + ] + assert api.calls[0]["cursor"] is None + assert set(result.item_evaluations) == {"obs-2", "obs-1", "obs-0"} + + +@pytest.mark.asyncio +async def test_observation_scores_can_be_duplicated_onto_trace(): + runner, _, client = make_runner([make_observation(1)]) + + result = await run(runner, _add_observation_scores_to_trace=True) + + assert result.total_scores_created == 2 + targets = [ + (call.kwargs.get("observation_id"), call.kwargs["trace_id"]) + for call in client.create_score.call_args_list + ] + assert targets == [("obs-1", "trace-1"), (None, "trace-1")] + + +@pytest.mark.asyncio +async def test_item_failures_and_evaluator_failures_are_tracked(): + observations = [make_observation(1), make_observation(2, trace_id=None)] + runner, _, client = make_runner(observations) + + def failing_evaluator(**kwargs): + raise RuntimeError("boom") + + def composite(*, evaluations, **kwargs): + return Evaluation(name="composite", value=len(evaluations)) + + result = await run( + runner, + evaluators=[length_evaluator, failing_evaluator], + composite_evaluator=composite, + ) + + assert result.total_items_processed == 1 + assert result.failed_item_ids == ["obs-2"] + assert result.error_summary == {"ValueError": 1} + assert result.total_evaluations_failed == 1 + assert result.total_composite_scores_created == 1 + assert [e.name for e in result.item_evaluations["obs-1"]] == [ + "length", + "composite", + ] + stats = {s.name: s for s in result.evaluator_stats} + assert stats["failing_evaluator"].failed_runs == 1 + assert stats["length_evaluator"].successful_runs == 1 + assert client.create_score.call_count == 2 + + +@pytest.mark.asyncio +async def test_async_mapper_and_evaluator_are_awaited(): + runner, _, _ = make_runner([make_observation(1)]) + + async def async_mapper(*, item): + return mapper(item=item) + + async def async_evaluator(*, output, **kwargs): + return [Evaluation(name="a", value=1), Evaluation(name="b", value=2)] + + result = await run(runner, mapper=async_mapper, evaluators=[async_evaluator]) + + assert result.total_scores_created == 2 From 638034e31085701db84a212e113180a6a1812de5 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 6 Oct 2026 14:27:34 +0000 Subject: [PATCH 11/25] test(batch-evaluation): rewrite e2e suite for v2 observations Co-authored-by: Hassieb Pakzad --- tests/e2e/test_batch_evaluation.py | 1292 +++++----------------------- 1 file changed, 225 insertions(+), 1067 deletions(-) diff --git a/tests/e2e/test_batch_evaluation.py b/tests/e2e/test_batch_evaluation.py index 0632b21b8..681dab1b5 100644 --- a/tests/e2e/test_batch_evaluation.py +++ b/tests/e2e/test_batch_evaluation.py @@ -1,1139 +1,297 @@ -"""Comprehensive tests for batch evaluation functionality. +"""End-to-end tests for run_batched_evaluation against a Langfuse server. -This test suite covers the run_batched_evaluation method which allows evaluating -traces, observations, and sessions fetched from Langfuse with mappers, evaluators, -and composite evaluators. +Every run is restricted to a corpus seeded by this module (filtered by a unique +tag), so assertions do not depend on other data in the project. Runner logic +that does not need a server is covered in tests/unit/test_batch_evaluation.py. """ -import asyncio -import time +import json +from dataclasses import dataclass +from typing import Any, List import pytest from langfuse import get_client, propagate_attributes -from langfuse.batch_evaluation import ( - BatchEvaluationResult, - BatchEvaluationResumeToken, - EvaluatorInputs, - EvaluatorStats, -) +from langfuse.api import ObservationV2 +from langfuse.batch_evaluation import EvaluatorInputs from langfuse.experiment import Evaluation from tests.support.utils import create_uuid, get_api, wait_for_result -# ============================================================================ -# FIXTURES & SETUP -# ============================================================================ +TRACE_COUNT = 4 -# pytestmark = pytest.mark.skip(reason="Github CI runner overwhelmed by score volume") +@dataclass +class Corpus: + tag: str + filter: str + trace_ids: List[str] + root_ids: List[str] + child_ids: List[str] -@pytest.fixture -def langfuse_client(): - """Get a Langfuse client for testing.""" - return get_client() - - -@pytest.fixture -def sample_trace_name(): - """Generate a unique trace name for filtering.""" - return f"batch-eval-test-{create_uuid()}" - - -def _seed_trace_corpus( - *, trace_count: int = 6, tag: str | None = None -) -> tuple[str, list[str]]: - langfuse_client = get_client() - corpus_tag = tag or f"batch-eval-seed-{create_uuid()}" - trace_names: list[str] = [] - - for index in range(trace_count): - trace_name = f"{corpus_tag}-trace-{index}" - trace_names.append(trace_name) - with langfuse_client.start_as_current_observation(name=trace_name) as span: - with propagate_attributes(tags=[corpus_tag]): - span.set_trace_io( - input=f"Seed input {index}", - output=f"Seed output {index}", - ) +def _tag_filter(tag: str) -> str: + return json.dumps( + [ + { + "type": "arrayOptions", + "column": "tags", + "operator": "any of", + "value": [tag], + } + ] + ) - langfuse_client.flush() - filter_json = f'[{{"type": "arrayOptions", "column": "tags", "operator": "any of", "value": ["{corpus_tag}"]}}]' +def _wait_for_observations(filter_json: str, expected_count: int) -> List[Any]: api = get_api(retry=False) - wait_for_result( - lambda: api.trace.list(filter=filter_json, limit=trace_count), - is_result_ready=lambda response: len(response.data) >= trace_count, + response = wait_for_result( + lambda: api.observations.get_many(filter=filter_json, limit=100), + is_result_ready=lambda r: len(r.data) >= expected_count, ) + return list(response.data) - return corpus_tag, trace_names - - -@pytest.fixture(scope="module", autouse=True) -def seeded_batch_evaluation_traces(): - _seed_trace_corpus() - -def simple_trace_mapper(*, item): - """Simple mapper for traces.""" +def _wait_for_scores(**kwargs: Any) -> List[Any]: + api = get_api(retry=False) + response = wait_for_result( + lambda: api.scores_v3.get_many_v3(fields="details,subject", **kwargs), + is_result_ready=lambda r: len(r.data) > 0, + ) + return list(response.data) + + +@pytest.fixture(scope="module") +def corpus() -> Corpus: + langfuse = get_client() + tag = f"batch-eval-{create_uuid()}" + trace_ids, root_ids, child_ids = [], [], [] + + for index in range(TRACE_COUNT): + with langfuse.start_as_current_observation(name=f"{tag}-root") as root: + with propagate_attributes(tags=[tag]): + with langfuse.start_as_current_observation( + as_type="generation", + name=f"{tag}-child", + input={"question": index}, + output=f"child answer {index}", + ) as child: + child_ids.append(child.id) + root.update(input={"question": index}, output=f"answer {index}") + trace_ids.append(root.trace_id) + root_ids.append(root.id) + + langfuse.flush() + _wait_for_observations(_tag_filter(tag), 2 * TRACE_COUNT) + + return Corpus( + tag=tag, + filter=_tag_filter(tag), + trace_ids=trace_ids, + root_ids=root_ids, + child_ids=child_ids, + ) + + +def io_mapper(*, item: ObservationV2) -> EvaluatorInputs: return EvaluatorInputs( - input=item.input if hasattr(item, "input") else None, - output=item.output if hasattr(item, "output") else None, - expected_output=None, - metadata={"trace_id": item.id}, + input=item.input, + output=item.output, + metadata={"trace_id": item.trace_id}, ) -def simple_evaluator(*, input, output, expected_output=None, metadata=None, **kwargs): - """Simple evaluator that returns a score based on output length.""" - if output is None: - return Evaluation(name="length_score", value=0.0, comment="No output") +def length_evaluator(*, output, **kwargs): + return Evaluation(name="length", value=float(len(output or ""))) - return Evaluation( - name="length_score", - value=float(len(str(output))) / 10.0, - comment=f"Length: {len(str(output))}", - ) +def test_observations_scope_evaluates_every_observation(corpus): + seen: List[ObservationV2] = [] -# ============================================================================ -# BASIC FUNCTIONALITY TESTS -# ============================================================================ + def recording_mapper(*, item): + seen.append(item) + return io_mapper(item=item) - -def test_run_batched_evaluation_on_observations_basic(langfuse_client): - """Test basic batch evaluation on traces.""" - result = langfuse_client.run_batched_evaluation( + result = get_client().run_batched_evaluation( scope="observations", - mapper=simple_trace_mapper, - evaluators=[simple_evaluator], - max_items=1, - verbose=True, - ) - - # Validate result structure - assert isinstance(result, BatchEvaluationResult) - assert result.total_items_fetched >= 0 - assert result.total_items_processed >= 0 - assert result.total_scores_created >= 0 - assert result.completed is True - assert isinstance(result.duration_seconds, float) - assert result.duration_seconds > 0 - - # Verify evaluator stats - assert len(result.evaluator_stats) == 1 - stats = result.evaluator_stats[0] - assert isinstance(stats, EvaluatorStats) - assert stats.name == "simple_evaluator" - - -def test_run_batched_evaluation_on_traces_basic(langfuse_client): - """Test basic batch evaluation on traces.""" - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[simple_evaluator], - max_items=5, - verbose=True, + mapper=recording_mapper, + evaluators=[length_evaluator], + filter=corpus.filter, ) - # Validate result structure - assert isinstance(result, BatchEvaluationResult) - assert result.total_items_fetched >= 0 - assert result.total_items_processed >= 0 - assert result.total_scores_created >= 0 assert result.completed is True - assert isinstance(result.duration_seconds, float) - assert result.duration_seconds > 0 - - # Verify evaluator stats - assert len(result.evaluator_stats) == 1 - stats = result.evaluator_stats[0] - assert isinstance(stats, EvaluatorStats) - assert stats.name == "simple_evaluator" + assert result.has_more_items is False + assert result.resume_token is None + assert result.total_items_fetched == 2 * TRACE_COUNT + assert result.total_items_processed == 2 * TRACE_COUNT + assert result.total_scores_created == 2 * TRACE_COUNT + assert set(result.item_evaluations) == set(corpus.root_ids + corpus.child_ids) + by_id = {item.id: item for item in seen} + child = by_id[corpus.child_ids[0]] + assert isinstance(child.input, str) + assert json.loads(child.input) == {"question": 0} + assert child.output == "child answer 0" -def test_batch_evaluation_with_filter(langfuse_client): - """Test batch evaluation with JSON filter.""" - # Create a trace with specific tag - unique_tag = f"test-filter-{create_uuid()}" - with langfuse_client.start_as_current_observation( - name=f"filtered-trace-{create_uuid()}" - ) as span: - with propagate_attributes(tags=[unique_tag]): - span.set_trace_io( - input="Filtered test", - output="Filtered output", - ) - langfuse_client.flush() - time.sleep(3) +def test_root_observations_scope_scores_traces(corpus): + score_name = f"root-score-{create_uuid()}" - # Filter format: array of filter conditions - filter_json = f'[{{"type": "arrayOptions", "column": "tags", "operator": "any of", "value": ["{unique_tag}"]}}]' + def root_evaluator(**kwargs): + return Evaluation(name=score_name, value=1.0) - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[simple_evaluator], - filter=filter_json, - verbose=True, + result = get_client().run_batched_evaluation( + scope="root_observations", + mapper=io_mapper, + evaluators=[root_evaluator], + filter=corpus.filter, ) - # Should only process the filtered trace - assert result.total_items_fetched >= 1 assert result.completed is True - - -def test_batch_evaluation_with_metadata(langfuse_client): - """Test that additional metadata is added to all scores.""" - - def metadata_checking_evaluator(*, input, output, metadata=None, **kwargs): - return Evaluation( - name="test_score", - value=1.0, - metadata={"evaluator_data": "test"}, - ) - - additional_metadata = { - "batch_run_id": "test-batch-123", - "evaluation_version": "v2.0", + assert result.total_items_processed == TRACE_COUNT + assert set(result.item_evaluations) == set(corpus.root_ids) + + scores = _wait_for_scores(trace_id=corpus.trace_ids[0], name=score_name) + assert len(scores) == 1 + assert scores[0].subject.kind == "trace" + assert scores[0].subject.id == corpus.trace_ids[0] + + +def test_observation_scores_are_attached_to_observations(corpus): + score_name = f"obs-score-{create_uuid()}" + child_filter = json.loads(corpus.filter) + [ + { + "type": "string", + "column": "id", + "operator": "=", + "value": corpus.child_ids[1], + } + ] + + def observation_evaluator(**kwargs): + return Evaluation(name=score_name, value=0.5, comment="ok") + + result = get_client().run_batched_evaluation( + scope="observations", + mapper=io_mapper, + evaluators=[observation_evaluator], + filter=json.dumps(child_filter), + metadata={"run": "e2e"}, + ) + + assert result.total_items_processed == 1 + + scores = _wait_for_scores( + trace_id=corpus.trace_ids[1], + observation_id=corpus.child_ids[1], + name=score_name, + ) + assert len(scores) == 1 + assert scores[0].subject.kind == "observation" + assert scores[0].subject.id == corpus.child_ids[1] + assert scores[0].subject.trace_id == corpus.trace_ids[1] + assert scores[0].comment == "ok" + assert scores[0].metadata == {"run": "e2e"} + + +def test_max_items_then_resume_covers_corpus_exactly_once(corpus): + langfuse = get_client() + run_kwargs: Any = { + "scope": "observations", + "mapper": io_mapper, + "evaluators": [length_evaluator], + "filter": corpus.filter, + "fetch_batch_size": 2, } - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[metadata_checking_evaluator], - metadata=additional_metadata, - max_items=2, - ) - - assert result.total_scores_created > 0 - - # Verify scores were created with merged metadata - langfuse_client.flush() - time.sleep(3) - - # Note: In a real test, you'd verify via API that metadata was merged - # For now, just verify the operation completed - assert result.completed is True - - -def test_result_structure_fields(langfuse_client): - """Test that BatchEvaluationResult has all expected fields.""" - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[simple_evaluator], - max_items=3, - ) - - # Check all result fields exist - assert hasattr(result, "total_items_fetched") - assert hasattr(result, "total_items_processed") - assert hasattr(result, "total_items_failed") - assert hasattr(result, "total_scores_created") - assert hasattr(result, "total_composite_scores_created") - assert hasattr(result, "total_evaluations_failed") - assert hasattr(result, "evaluator_stats") - assert hasattr(result, "resume_token") - assert hasattr(result, "completed") - assert hasattr(result, "duration_seconds") - assert hasattr(result, "failed_item_ids") - assert hasattr(result, "error_summary") - assert hasattr(result, "has_more_items") - assert hasattr(result, "item_evaluations") - - # Check types - assert isinstance(result.evaluator_stats, list) - assert isinstance(result.failed_item_ids, list) - assert isinstance(result.error_summary, dict) - assert isinstance(result.completed, bool) - assert isinstance(result.has_more_items, bool) - assert isinstance(result.item_evaluations, dict) - - -# ============================================================================ -# MAPPER FUNCTION TESTS -# ============================================================================ - - -def test_simple_mapper(langfuse_client): - """Test basic mapper functionality.""" - - def custom_mapper(*, item): - return EvaluatorInputs( - input=item.input if hasattr(item, "input") else "no input", - output=item.output if hasattr(item, "output") else "no output", - expected_output=None, - metadata={"custom_field": "test_value"}, - ) - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=custom_mapper, - evaluators=[simple_evaluator], - max_items=2, - ) - - assert result.total_items_processed > 0 - - -@pytest.mark.asyncio -async def test_async_mapper(langfuse_client): - """Test that async mappers work correctly.""" - - async def async_mapper(*, item): - await asyncio.sleep(0.01) # Simulate async work - return EvaluatorInputs( - input=item.input if hasattr(item, "input") else None, - output=item.output if hasattr(item, "output") else None, - expected_output=None, - metadata={"async": True}, - ) - - # Note: run_batched_evaluation is synchronous but handles async mappers - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=async_mapper, - evaluators=[simple_evaluator], - max_items=2, - ) - - assert result.total_items_processed > 0 - - -def test_mapper_failure_handling(langfuse_client): - """Test that mapper failures cause items to be skipped.""" - - def failing_mapper(*, item): - raise ValueError("Intentional mapper failure") - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=failing_mapper, - evaluators=[simple_evaluator], - max_items=3, - ) - - # All items should fail due to mapper failures - assert result.total_items_failed > 0 - assert len(result.failed_item_ids) > 0 - assert "ValueError" in result.error_summary or "Exception" in result.error_summary - - -def test_mapper_with_missing_fields(langfuse_client): - """Test mapper handles traces with missing fields gracefully.""" - - def robust_mapper(*, item): - # Handle missing fields with defaults - input_val = getattr(item, "input", None) or "default_input" - output_val = getattr(item, "output", None) or "default_output" - - return EvaluatorInputs( - input=input_val, - output=output_val, - expected_output=None, - metadata={}, - ) - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=robust_mapper, - evaluators=[simple_evaluator], - max_items=2, - ) - - assert result.total_items_processed > 0 - - -# ============================================================================ -# EVALUATOR TESTS -# ============================================================================ - - -def test_single_evaluator(langfuse_client): - """Test with a single evaluator.""" - - def quality_evaluator(*, input, output, **kwargs): - return Evaluation(name="quality", value=0.85, comment="High quality") - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[quality_evaluator], - max_items=2, - ) - - assert result.total_scores_created > 0 - assert len(result.evaluator_stats) == 1 - assert result.evaluator_stats[0].name == "quality_evaluator" - - -def test_multiple_evaluators(langfuse_client): - """Test with multiple evaluators running in parallel.""" - - def accuracy_evaluator(*, input, output, **kwargs): - return Evaluation(name="accuracy", value=0.9) - - def relevance_evaluator(*, input, output, **kwargs): - return Evaluation(name="relevance", value=0.8) - - def safety_evaluator(*, input, output, **kwargs): - return Evaluation(name="safety", value=1.0) - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[accuracy_evaluator, relevance_evaluator, safety_evaluator], - max_items=2, - ) - - # Should have 3 evaluators - assert len(result.evaluator_stats) == 3 - assert result.total_scores_created >= result.total_items_processed * 3 - - -@pytest.mark.asyncio -async def test_async_evaluator(langfuse_client): - """Test that async evaluators work correctly.""" - - async def async_evaluator(*, input, output, **kwargs): - await asyncio.sleep(0.01) # Simulate async work - return Evaluation(name="async_score", value=0.75) - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[async_evaluator], - max_items=2, - ) - - assert result.total_scores_created > 0 - - -def test_evaluator_returning_list(langfuse_client): - """Test evaluator that returns multiple Evaluations.""" - - def multi_score_evaluator(*, input, output, **kwargs): - return [ - Evaluation(name="score_1", value=0.8), - Evaluation(name="score_2", value=0.9), - Evaluation(name="score_3", value=0.7), - ] - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[multi_score_evaluator], - max_items=2, - ) - - # Should create 3 scores per item - assert result.total_scores_created >= result.total_items_processed * 3 - - -def test_evaluator_failure_statistics(langfuse_client): - """Test that evaluator failures are tracked in statistics.""" - - def working_evaluator(*, input, output, **kwargs): - return Evaluation(name="working", value=1.0) - - def failing_evaluator(*, input, output, **kwargs): - raise RuntimeError("Intentional evaluator failure") - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[working_evaluator, failing_evaluator], - max_items=3, - ) - - # Verify evaluator stats - assert len(result.evaluator_stats) == 2 - - working_stats = next( - s for s in result.evaluator_stats if s.name == "working_evaluator" - ) - assert working_stats.successful_runs > 0 - assert working_stats.failed_runs == 0 - - failing_stats = next( - s for s in result.evaluator_stats if s.name == "failing_evaluator" - ) - assert failing_stats.failed_runs > 0 - assert failing_stats.successful_runs == 0 - - # Total evaluations failed should be tracked - assert result.total_evaluations_failed > 0 - + first = langfuse.run_batched_evaluation(max_items=3, **run_kwargs) -def test_mixed_sync_async_evaluators(langfuse_client): - """Test mixing synchronous and asynchronous evaluators.""" + assert first.total_items_fetched == 3 + assert first.has_more_items is True + assert first.resume_token is not None + assert first.resume_token.cursor is not None - def sync_evaluator(*, input, output, **kwargs): - return Evaluation(name="sync_score", value=0.8) - - async def async_evaluator(*, input, output, **kwargs): - await asyncio.sleep(0.01) - return Evaluation(name="async_score", value=0.9) - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[sync_evaluator, async_evaluator], - max_items=2, - ) - - assert len(result.evaluator_stats) == 2 - assert result.total_scores_created >= result.total_items_processed * 2 - - -# ============================================================================ -# COMPOSITE EVALUATOR TESTS -# ============================================================================ - - -def test_composite_evaluator_weighted_average(langfuse_client): - """Test composite evaluator that computes weighted average.""" - - def accuracy_evaluator(*, input, output, **kwargs): - return Evaluation(name="accuracy", value=0.8) - - def relevance_evaluator(*, input, output, **kwargs): - return Evaluation(name="relevance", value=0.9) - - def composite_evaluator(*, input, output, expected_output, metadata, evaluations): - weights = {"accuracy": 0.6, "relevance": 0.4} - total = sum( - e.value * weights.get(e.name, 0) - for e in evaluations - if isinstance(e.value, (int, float)) - ) - - return Evaluation( - name="composite_score", - value=total, - comment=f"Weighted average of {len(evaluations)} metrics", - ) - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[accuracy_evaluator, relevance_evaluator], - composite_evaluator=composite_evaluator, - max_items=2, - ) - - # Should have both regular and composite scores - assert result.total_scores_created > 0 - assert result.total_composite_scores_created > 0 - assert result.total_scores_created > result.total_composite_scores_created - - -def test_composite_evaluator_pass_fail(langfuse_client): - """Test composite evaluator that implements pass/fail logic.""" - - def metric1_evaluator(*, input, output, **kwargs): - return Evaluation(name="metric1", value=0.9) - - def metric2_evaluator(*, input, output, **kwargs): - return Evaluation(name="metric2", value=0.7) - - def pass_fail_composite(*, input, output, expected_output, metadata, evaluations): - thresholds = {"metric1": 0.8, "metric2": 0.6} - - passes = all( - e.value >= thresholds.get(e.name, 0) - for e in evaluations - if isinstance(e.value, (int, float)) - ) - - return Evaluation( - name="passes_all_checks", - value=1.0 if passes else 0.0, - comment="All checks passed" if passes else "Some checks failed", - ) - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[metric1_evaluator, metric2_evaluator], - composite_evaluator=pass_fail_composite, - max_items=2, - ) - - assert result.total_composite_scores_created > 0 - - -@pytest.mark.asyncio -async def test_async_composite_evaluator(langfuse_client): - """Test async composite evaluator.""" - - def evaluator1(*, input, output, **kwargs): - return Evaluation(name="eval1", value=0.8) - - async def async_composite(*, input, output, expected_output, metadata, evaluations): - await asyncio.sleep(0.01) # Simulate async processing - avg = sum( - e.value for e in evaluations if isinstance(e.value, (int, float)) - ) / len(evaluations) - return Evaluation(name="async_composite", value=avg) - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[evaluator1], - composite_evaluator=async_composite, - max_items=2, + second = langfuse.run_batched_evaluation( + resume_from=first.resume_token, **run_kwargs ) - assert result.total_composite_scores_created > 0 - - -def test_composite_evaluator_with_no_evaluations(langfuse_client): - """Test composite evaluator when no evaluations are present.""" - - def always_failing_evaluator(*, input, output, **kwargs): - raise Exception("Always fails") - - def composite_evaluator(*, input, output, expected_output, metadata, evaluations): - # Should not be called if no evaluations succeed - return Evaluation(name="composite", value=0.0) - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[always_failing_evaluator], - composite_evaluator=composite_evaluator, - max_items=2, + assert second.completed is True + assert second.resume_token is None + assert set(first.item_evaluations).isdisjoint(second.item_evaluations) + assert set(first.item_evaluations) | set(second.item_evaluations) == set( + corpus.root_ids + corpus.child_ids ) - # Composite evaluator should not create scores if no evaluations - assert result.total_composite_scores_created == 0 - -def test_composite_evaluator_failure_handling(langfuse_client): - """Test that composite evaluator failures are handled gracefully.""" +def test_fields_control_populated_field_groups(corpus): + seen: List[ObservationV2] = [] - def evaluator1(*, input, output, **kwargs): - return Evaluation(name="eval1", value=0.8) + def recording_mapper(*, item): + seen.append(item) + return EvaluatorInputs(input=None, output=None) - def failing_composite(*, input, output, expected_output, metadata, evaluations): - raise ValueError("Composite evaluator failed") - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[evaluator1], - composite_evaluator=failing_composite, - max_items=2, - ) - - # Regular scores should still be created - assert result.total_scores_created > 0 - # But no composite scores - assert result.total_composite_scores_created == 0 - - -# ============================================================================ -# ERROR HANDLING TESTS -# ============================================================================ - - -def test_mapper_failure_skips_item(langfuse_client): - """Test that mapper failure causes item to be skipped.""" - - call_count = {"count": 0} - - def sometimes_failing_mapper(*, item): - call_count["count"] += 1 - if call_count["count"] % 2 == 0: - raise Exception("Mapper failed") - return simple_trace_mapper(item=item) - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=sometimes_failing_mapper, - evaluators=[simple_evaluator], - max_items=4, - ) - - # Some items should fail, some should succeed - assert result.total_items_failed > 0 - assert result.total_items_processed > 0 - - -def test_evaluator_failure_continues(langfuse_client): - """Test that one evaluator failing doesn't stop others.""" - - def working_evaluator1(*, input, output, **kwargs): - return Evaluation(name="working1", value=0.8) - - def failing_evaluator(*, input, output, **kwargs): - raise Exception("Evaluator failed") - - def working_evaluator2(*, input, output, **kwargs): - return Evaluation(name="working2", value=0.9) - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[working_evaluator1, failing_evaluator, working_evaluator2], - max_items=2, - ) - - # Working evaluators should still create scores - assert result.total_scores_created >= result.total_items_processed * 2 - - # Failing evaluator should be tracked - failing_stats = next( - s for s in result.evaluator_stats if s.name == "failing_evaluator" - ) - assert failing_stats.failed_runs > 0 - - -def test_all_evaluators_fail(langfuse_client): - """Test when all evaluators fail but item is still processed.""" - - def failing_evaluator1(*, input, output, **kwargs): - raise Exception("Failed 1") - - def failing_evaluator2(*, input, output, **kwargs): - raise Exception("Failed 2") - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[failing_evaluator1, failing_evaluator2], - max_items=2, - ) - - # Items should be processed even if all evaluators fail - assert result.total_items_processed > 0 - # But no scores created - assert result.total_scores_created == 0 - # All evaluations failed - assert result.total_evaluations_failed > 0 - - -# ============================================================================ -# EDGE CASES TESTS -# ============================================================================ - - -def test_empty_results_handling(langfuse_client): - """Test batch evaluation when filter returns no items.""" - nonexistent_name = f"nonexistent-trace-{create_uuid()}" - nonexistent_filter = f'[{{"type": "string", "column": "name", "operator": "=", "value": "{nonexistent_name}"}}]' - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[simple_evaluator], - filter=nonexistent_filter, - ) - - assert result.total_items_fetched == 0 - assert result.total_items_processed == 0 - assert result.total_scores_created == 0 - assert result.completed is True - assert result.has_more_items is False - - -def test_max_items_zero(langfuse_client): - """Test with max_items=0 (should process no items).""" - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[simple_evaluator], - max_items=0, - ) - - assert result.total_items_fetched == 0 - assert result.total_items_processed == 0 - - -def test_evaluation_value_type_conversions(langfuse_client): - """Test that different evaluation value types are handled correctly.""" - - def multi_type_evaluator(*, input, output, **kwargs): - return [ - Evaluation(name="int_score", value=5), # int - Evaluation(name="float_score", value=0.85), # float - Evaluation(name="bool_score", value=True), # bool - Evaluation(name="none_score", value=None), # None - ] - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[multi_type_evaluator], + get_client().run_batched_evaluation( + scope="root_observations", + mapper=recording_mapper, + evaluators=[length_evaluator], + filter=corpus.filter, + fields="core,basic", max_items=1, ) - # All value types should be converted and scores created - assert result.total_scores_created >= 4 - - -# ============================================================================ -# PAGINATION TESTS -# ============================================================================ - - -def test_pagination_with_max_items(langfuse_client): - """Test that max_items limit is respected.""" - # Create more traces to ensure we have enough data - for i in range(10): - with langfuse_client.start_as_current_observation( - name=f"pagination-test-{create_uuid()}" - ) as span: - with propagate_attributes(tags=["pagination_test"]): - span.set_trace_io( - input=f"Input {i}", - output=f"Output {i}", - ) - - langfuse_client.flush() - time.sleep(3) - - filter_json = '[{"type": "arrayOptions", "column": "tags", "operator": "any of", "value": ["pagination_test"]}]' - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[simple_evaluator], - filter=filter_json, - max_items=5, - fetch_batch_size=2, - ) - - # Should not exceed max_items - assert result.total_items_processed <= 5 - - -def test_has_more_items_flag(langfuse_client): - """Test that has_more_items flag is set correctly when max_items is reached.""" - # Create enough traces to exceed max_items - batch_tag = f"batch-test-{create_uuid()}" - for i in range(15): - with langfuse_client.start_as_current_observation( - name=f"more-items-test-{i}" - ) as span: - with propagate_attributes(tags=[batch_tag]): - span.set_trace_io( - input=f"Input {i}", - output=f"Output {i}", - ) - - langfuse_client.flush() - time.sleep(3) - - filter_json = f'[{{"type": "arrayOptions", "column": "tags", "operator": "any of", "value": ["{batch_tag}"]}}]' - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[simple_evaluator], - filter=filter_json, - max_items=5, - fetch_batch_size=2, - ) - - # has_more_items should be True if we hit the limit - if result.total_items_fetched >= 5: - assert result.has_more_items is True - - -def test_fetch_batch_size_parameter(langfuse_client): - """Test that different fetch_batch_size values work correctly.""" - for batch_size in [1, 5, 10]: - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[simple_evaluator], - max_items=3, - fetch_batch_size=batch_size, - ) - - # Should complete regardless of batch size - assert result.completed is True or result.total_items_processed > 0 + assert len(seen) == 1 + assert seen[0].input is None + assert seen[0].is_root_observation is True -# ============================================================================ -# RESUME FUNCTIONALITY TESTS -# ============================================================================ +def test_composite_evaluator_and_failures(corpus): + def failing_evaluator(**kwargs): + raise RuntimeError("intentional") + def composite(*, evaluations, **kwargs): + return Evaluation(name="composite", value=float(len(evaluations))) -def test_resume_token_structure(langfuse_client): - """Test that BatchEvaluationResumeToken has correct structure.""" - resume_token = BatchEvaluationResumeToken( - scope="traces", - filter='{"test": "filter"}', - last_processed_timestamp="2024-01-01T00:00:00Z", - last_processed_id="trace-123", - items_processed=10, + result = get_client().run_batched_evaluation( + scope="root_observations", + mapper=io_mapper, + evaluators=[length_evaluator, failing_evaluator], + composite_evaluator=composite, + filter=corpus.filter, ) - assert resume_token.scope == "traces" - assert resume_token.filter == '{"test": "filter"}' - assert resume_token.last_processed_timestamp == "2024-01-01T00:00:00Z" - assert resume_token.last_processed_id == "trace-123" - assert resume_token.items_processed == 10 - - -# ============================================================================ -# CONCURRENCY TESTS -# ============================================================================ - - -def test_max_concurrency_parameter(langfuse_client): - """Test that max_concurrency parameter works correctly.""" - for concurrency in [1, 5, 10]: - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[simple_evaluator], - max_items=3, - max_concurrency=concurrency, - ) + assert result.total_items_processed == TRACE_COUNT + assert result.total_scores_created == TRACE_COUNT + assert result.total_composite_scores_created == TRACE_COUNT + assert result.total_evaluations_failed == TRACE_COUNT + stats = {s.name: s for s in result.evaluator_stats} + assert stats["failing_evaluator"].failed_runs == TRACE_COUNT - # Should complete regardless of concurrency - assert result.completed is True or result.total_items_processed > 0 - - -# ============================================================================ -# STATISTICS TESTS -# ============================================================================ - - -def test_evaluator_stats_structure(langfuse_client): - """Test that EvaluatorStats has correct structure.""" - - def test_evaluator(*, input, output, **kwargs): - return Evaluation(name="test", value=1.0) - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[test_evaluator], - max_items=2, - ) - - assert len(result.evaluator_stats) == 1 - stats = result.evaluator_stats[0] - - # Check all fields exist - assert hasattr(stats, "name") - assert hasattr(stats, "total_runs") - assert hasattr(stats, "successful_runs") - assert hasattr(stats, "failed_runs") - assert hasattr(stats, "total_scores_created") - - # Check values - assert stats.name == "test_evaluator" - assert stats.total_runs == result.total_items_processed - assert stats.successful_runs == result.total_items_processed - assert stats.failed_runs == 0 - - -def test_evaluator_stats_tracking(langfuse_client): - """Test that evaluator statistics are tracked correctly.""" - - call_count = {"count": 0} - - def sometimes_failing_evaluator(*, input, output, **kwargs): - call_count["count"] += 1 - if call_count["count"] % 2 == 0: - raise Exception("Failed") - return Evaluation(name="test", value=1.0) - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[sometimes_failing_evaluator], - max_items=4, - ) - - stats = result.evaluator_stats[0] - assert stats.total_runs == result.total_items_processed - assert stats.successful_runs > 0 - assert stats.failed_runs > 0 - assert stats.successful_runs + stats.failed_runs == stats.total_runs - - -def test_error_summary_aggregation(langfuse_client): - """Test that error types are aggregated correctly in error_summary.""" - - def failing_mapper(*, item): - raise ValueError("Mapper error") - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=failing_mapper, - evaluators=[simple_evaluator], - max_items=3, - ) - - # Error summary should contain the error type - assert len(result.error_summary) > 0 - assert any("Error" in key for key in result.error_summary.keys()) - - -def test_failed_item_ids_collected(langfuse_client): - """Test that failed item IDs are collected.""" +def test_mapper_failures_are_reported_per_item(corpus): def failing_mapper(*, item): - raise Exception("Failed") + raise ValueError("intentional") - result = langfuse_client.run_batched_evaluation( - scope="traces", + result = get_client().run_batched_evaluation( + scope="root_observations", mapper=failing_mapper, - evaluators=[simple_evaluator], - max_items=3, - ) - - assert len(result.failed_item_ids) > 0 - # Each failed ID should be a string - assert all(isinstance(item_id, str) for item_id in result.failed_item_ids) - - -# ============================================================================ -# PERFORMANCE TESTS -# ============================================================================ - - -def test_duration_tracking(langfuse_client): - """Test that duration is tracked correctly.""" - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[simple_evaluator], - max_items=2, - ) - - assert result.duration_seconds > 0 - assert result.duration_seconds < 60 # Should complete quickly for small batch - - -def test_verbose_logging(langfuse_client): - """Test that verbose=True doesn't cause errors.""" - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[simple_evaluator], - max_items=2, - verbose=True, # Should log progress + evaluators=[length_evaluator], + filter=corpus.filter, ) assert result.completed is True + assert result.total_items_failed == TRACE_COUNT + assert set(result.failed_item_ids) == set(corpus.root_ids) + assert result.error_summary == {"ValueError": TRACE_COUNT} -# ============================================================================ -# ITEM EVALUATIONS TESTS -# ============================================================================ - - -def test_item_evaluations_basic(langfuse_client): - """Test that item_evaluations dict contains correct structure.""" - - def test_evaluator(*, input, output, **kwargs): - return Evaluation(name="test_metric", value=0.5) - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[test_evaluator], - max_items=3, - ) - - # Check that item_evaluations is a dict - assert isinstance(result.item_evaluations, dict) - - # Should have evaluations for each processed item - assert len(result.item_evaluations) == result.total_items_processed - - # Each entry should be a list of Evaluation objects - for item_id, evaluations in result.item_evaluations.items(): - assert isinstance(item_id, str) - assert isinstance(evaluations, list) - assert all(isinstance(e, Evaluation) for e in evaluations) - # Should have one evaluation per evaluator - assert len(evaluations) == 1 - assert evaluations[0].name == "test_metric" - - -def test_item_evaluations_multiple_evaluators(langfuse_client): - """Test item_evaluations with multiple evaluators.""" - - def accuracy_evaluator(*, input, output, **kwargs): - return Evaluation(name="accuracy", value=0.8) - - def relevance_evaluator(*, input, output, **kwargs): - return Evaluation(name="relevance", value=0.9) - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[accuracy_evaluator, relevance_evaluator], - max_items=2, - ) - - # Check structure - assert len(result.item_evaluations) == result.total_items_processed - - # Each item should have evaluations from both evaluators - for item_id, evaluations in result.item_evaluations.items(): - assert len(evaluations) == 2 - eval_names = {e.name for e in evaluations} - assert eval_names == {"accuracy", "relevance"} - - -def test_item_evaluations_with_composite(langfuse_client): - """Test that item_evaluations includes composite evaluations.""" - - def base_evaluator(*, input, output, **kwargs): - return Evaluation(name="base_score", value=0.7) - - def composite_evaluator(*, input, output, expected_output, metadata, evaluations): - return Evaluation( - name="composite_score", - value=sum( - e.value for e in evaluations if isinstance(e.value, (int, float)) - ), - ) - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=simple_trace_mapper, - evaluators=[base_evaluator], - composite_evaluator=composite_evaluator, - max_items=2, - ) - - # Each item should have both base and composite evaluations - for item_id, evaluations in result.item_evaluations.items(): - assert len(evaluations) == 2 - eval_names = {e.name for e in evaluations} - assert eval_names == {"base_score", "composite_score"} - - # Verify composite scores were created - assert result.total_composite_scores_created > 0 - - -def test_item_evaluations_empty_on_failure(langfuse_client): - """Test that failed items don't appear in item_evaluations.""" - - def failing_mapper(*, item): - raise Exception("Mapper failed") - - result = langfuse_client.run_batched_evaluation( - scope="traces", - mapper=failing_mapper, - evaluators=[simple_evaluator], - max_items=3, +def test_filter_without_matches_completes_empty(): + result = get_client().run_batched_evaluation( + scope="observations", + mapper=io_mapper, + evaluators=[length_evaluator], + filter=_tag_filter(f"nonexistent-{create_uuid()}"), ) - # All items failed, so item_evaluations should be empty - assert len(result.item_evaluations) == 0 - assert result.total_items_failed > 0 + assert result.completed is True + assert result.total_items_fetched == 0 + assert result.has_more_items is False + assert result.resume_token is None From 5fc61473e334fcbcf9076c94a7336b180c2af238 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 6 Oct 2026 15:02:16 +0000 Subject: [PATCH 12/25] fix(tracing): serialize propagated metadata like JSON.stringify Use compact separators, keep non-ASCII characters, and send None as "null", so the Python and JS SDKs emit byte-identical propagated metadata values. Co-authored-by: Hassieb Pakzad --- langfuse/_client/propagation.py | 24 ++++++++++++++++------- tests/unit/test_propagate_attributes.py | 26 ++++++++----------------- 2 files changed, 25 insertions(+), 25 deletions(-) diff --git a/langfuse/_client/propagation.py b/langfuse/_client/propagation.py index 1746554b8..18bab9799 100644 --- a/langfuse/_client/propagation.py +++ b/langfuse/_client/propagation.py @@ -5,6 +5,7 @@ propagate to all child spans within the context. """ +import json import re from typing import ( Any, @@ -38,8 +39,9 @@ _agnosticcontextmanager, ) -from langfuse._client.attributes import LangfuseOtelSpanAttributes, _serialize +from langfuse._client.attributes import LangfuseOtelSpanAttributes from langfuse._client.constants import LANGFUSE_SDK_EXPERIMENT_ENVIRONMENT +from langfuse._utils.serializer import EventSerializer from langfuse.logger import langfuse_logger from langfuse.model import PromptClient @@ -282,8 +284,9 @@ def propagate_attributes( trace_name) must be strings ≤200 characters. Environment must also match Langfuse's environment format: lowercase alphanumeric with optional hyphens or underscores, must be ≤40 characters, and it must not start with "langfuse". Non-string - metadata values are JSON-serialized before the 200 character limit is - applied, and None values are dropped. + metadata values are serialized like JavaScript's `JSON.stringify` + (compact separators, non-ASCII kept as is, None becomes "null") + before the 200 character limit is applied. Invalid values will be dropped with a warning logged. - **OpenTelemetry**: This uses OpenTelemetry context propagation under the hood, making it compatible with other OTel-instrumented libraries. @@ -394,10 +397,7 @@ def _propagate_attributes( validated_metadata: Dict[str, str] = {} for key, value in metadata_value.items(): - serialized_value = _serialize(value) - - if serialized_value is None: - continue + serialized_value = _serialize_propagated_metadata_value(value) if _validate_string_value( value=serialized_value, key=f"{metadata_key}.{key}" @@ -647,6 +647,16 @@ def _validate_propagated_value( return value +def _serialize_propagated_metadata_value(value: Any) -> str: + # Must match JSON.stringify in the JS SDK so both SDKs emit identical values. + if isinstance(value, str): + return value + + return json.dumps( + value, cls=EventSerializer, separators=(",", ":"), ensure_ascii=False + ) + + def _validate_string_value(*, value: str, key: str) -> bool: if not isinstance(value, str): langfuse_logger.warning( # type: ignore diff --git a/tests/unit/test_propagate_attributes.py b/tests/unit/test_propagate_attributes.py index c49a4e956..17b6a95a7 100644 --- a/tests/unit/test_propagate_attributes.py +++ b/tests/unit/test_propagate_attributes.py @@ -474,16 +474,21 @@ def test_non_string_metadata_values_json_serialized( "max_search_results": 5, "is_cached": True, "ratio": 0.5, - "config": {"model": "gpt-4o"}, + "config": {"model": "gpt-4o", "nested": {"b": [1, None], "a": "ü"}}, + "label": ["Läufe", "🚀"], + "empty": None, } + # Byte-identical to JSON.stringify in the JS SDK for the same values. expected = { "langgraph_step": "1", "langgraph_triggers": '["branch:agent"]', - "langgraph_path": '["root", "agent"]', + "langgraph_path": '["root","agent"]', "max_search_results": "5", "is_cached": "true", "ratio": "0.5", - "config": '{"model": "gpt-4o"}', + "config": '{"model":"gpt-4o","nested":{"b":[1,null],"a":"ü"}}', + "label": '["Läufe","🚀"]', + "empty": "null", } with langfuse_client.start_as_current_observation(name="parent-span"): @@ -502,21 +507,6 @@ def test_non_string_metadata_values_json_serialized( assert "value is not a string. Dropping value." not in caplog.text - def test_none_metadata_values_dropped(self, langfuse_client, memory_exporter): - """Verify None metadata values are dropped instead of sent as 'None'.""" - with langfuse_client.start_as_current_observation(name="parent-span"): - with propagate_attributes(metadata={"kept": "yes", "empty": None}): - child = langfuse_client.start_observation(name="child-span") - child.end() - - child_span = self.get_span_by_name(memory_exporter, "child-span") - self.verify_span_attribute( - child_span, f"{LangfuseOtelSpanAttributes.TRACE_METADATA}.kept", "yes" - ) - self.verify_missing_attribute( - child_span, f"{LangfuseOtelSpanAttributes.TRACE_METADATA}.empty" - ) - def test_mixed_valid_invalid_metadata(self, langfuse_client, memory_exporter): """Verify mixed valid/invalid metadata - valid entries kept, invalid dropped.""" with langfuse_client.start_as_current_observation(name="parent-span"): From 4cffec571da70bef5b3dc32cc86cdc3c0c5274a9 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 7 Oct 2026 08:39:58 +0000 Subject: [PATCH 13/25] fix(observe): keep detecting legacy-marked coroutine functions inspect.iscoroutinefunction ignores asyncio's _is_coroutine marker, which asgiref's markcoroutinefunction sets on Python < 3.12. Accept both so such functions keep the async wrapper and their observation ends after the coroutine runs. Co-authored-by: Hassieb Pakzad --- langfuse/_client/observe.py | 13 ++++++++++++- tests/unit/test_observe.py | 24 ++++++++++++++++++++++++ 2 files changed, 36 insertions(+), 1 deletion(-) diff --git a/langfuse/_client/observe.py b/langfuse/_client/observe.py index 9d28d8156..5aa88ce2e 100644 --- a/langfuse/_client/observe.py +++ b/langfuse/_client/observe.py @@ -48,6 +48,17 @@ P = ParamSpec("P") R = TypeVar("R") +# Set by asgiref's markcoroutinefunction on Python < 3.12, which +# inspect.iscoroutinefunction does not recognize. +_ASYNCIO_COROUTINE_MARKER = getattr(asyncio.coroutines, "_is_coroutine", None) + + +def _is_coroutine_function(func: Any) -> bool: + return inspect.iscoroutinefunction(func) or ( + _ASYNCIO_COROUTINE_MARKER is not None + and getattr(func, "_is_coroutine", None) is _ASYNCIO_COROUTINE_MARKER + ) + class LangfuseDecorator: """Implementation of the @observe decorator for seamless Langfuse tracing integration. @@ -214,7 +225,7 @@ def decorator(func: F) -> F: capture_output=should_capture_output, transform_to_string=transform_to_string, ) - if inspect.iscoroutinefunction(func) + if _is_coroutine_function(func) else self._sync_observe( func, name=name, diff --git a/tests/unit/test_observe.py b/tests/unit/test_observe.py index 865779591..21ff1418d 100644 --- a/tests/unit/test_observe.py +++ b/tests/unit/test_observe.py @@ -499,3 +499,27 @@ async def generator() -> AsyncGenerator[str, None]: assert span.ended == 1 assert span.updates == [] + + +@pytest.mark.asyncio +async def test_observe_treats_legacy_marked_coroutine_function_as_async( + langfuse_memory_client: Any, memory_exporter: Any +) -> None: + async def work() -> str: + await asyncio.sleep(0) + return "done" + + def marked() -> Any: + return work() + + # The marker asgiref's markcoroutinefunction sets on Python < 3.12. + cast(Any, marked)._is_coroutine = asyncio.coroutines._is_coroutine # type: ignore[attr-defined] + + observed = observe(name="marked")(marked) + + assert await observed() == "done" + + langfuse_memory_client.flush() + + span = _finished_spans_by_name(memory_exporter, "marked")[0] + assert span.attributes[LangfuseOtelSpanAttributes.OBSERVATION_OUTPUT] == "done" From 846cc7fe35303026dc76745981ffaace198b0dcb Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 7 Oct 2026 08:43:13 +0000 Subject: [PATCH 14/25] test(otel): pin that a mixed-case ingestion-version override wins Co-authored-by: Hassieb Pakzad --- tests/unit/test_additional_headers_simple.py | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) diff --git a/tests/unit/test_additional_headers_simple.py b/tests/unit/test_additional_headers_simple.py index fd7764fd6..d05475a3c 100644 --- a/tests/unit/test_additional_headers_simple.py +++ b/tests/unit/test_additional_headers_simple.py @@ -227,6 +227,25 @@ def test_span_processor_additional_headers_override_ingestion_version(self): assert exporter._client._headers["x-langfuse-ingestion-version"] == "3" + def test_span_processor_mixed_case_override_replaces_ingestion_version(self): + """Test that a differently cased override replaces the default header.""" + from langfuse._client.span_processor import LangfuseSpanProcessor + + processor = LangfuseSpanProcessor( + public_key="test-public-key", + secret_key="test-secret-key", + base_url="https://mock-host.com", + additional_headers={"X-Langfuse-Ingestion-Version": "3"}, + ) + + ingestion_headers = { + key: value + for key, value in processor.span_exporter._client._headers.items() + if key.lower() == "x-langfuse-ingestion-version" + } + + assert ingestion_headers == {"x-langfuse-ingestion-version": "3"} + def test_span_processor_uses_custom_span_exporter_when_provided(self): """Test that a custom exporter bypasses the default OTLP exporter construction.""" from langfuse._client.span_processor import LangfuseSpanProcessor From 14da6114f3dabe9610e0c6c236dd65d540ee0802 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 7 Oct 2026 08:43:48 +0000 Subject: [PATCH 15/25] fix(tracing): keep the exact digits of large integer metadata values Co-authored-by: Hassieb Pakzad --- langfuse/_client/propagation.py | 4 ++++ tests/unit/test_propagate_attributes.py | 16 ++++++++++++++++ 2 files changed, 20 insertions(+) diff --git a/langfuse/_client/propagation.py b/langfuse/_client/propagation.py index 18bab9799..2eecc8647 100644 --- a/langfuse/_client/propagation.py +++ b/langfuse/_client/propagation.py @@ -652,6 +652,10 @@ def _serialize_propagated_metadata_value(value: Any) -> str: if isinstance(value, str): return value + # EventSerializer quotes ints outside JS's safe range; keep their exact digits. + if isinstance(value, int) and not isinstance(value, bool): + return str(value) + return json.dumps( value, cls=EventSerializer, separators=(",", ":"), ensure_ascii=False ) diff --git a/tests/unit/test_propagate_attributes.py b/tests/unit/test_propagate_attributes.py index 17b6a95a7..bd6003e47 100644 --- a/tests/unit/test_propagate_attributes.py +++ b/tests/unit/test_propagate_attributes.py @@ -507,6 +507,22 @@ def test_non_string_metadata_values_json_serialized( assert "value is not a string. Dropping value." not in caplog.text + def test_large_integer_metadata_keeps_its_digits( + self, langfuse_client, memory_exporter + ): + """Verify integers beyond JS's safe range are sent as plain digits.""" + with langfuse_client.start_as_current_observation(name="parent-span"): + with propagate_attributes(metadata={"snowflake_id": 9007199254740993}): + child = langfuse_client.start_observation(name="child-span") + child.end() + + child_span = self.get_span_by_name(memory_exporter, "child-span") + self.verify_span_attribute( + child_span, + f"{LangfuseOtelSpanAttributes.TRACE_METADATA}.snowflake_id", + "9007199254740993", + ) + def test_mixed_valid_invalid_metadata(self, langfuse_client, memory_exporter): """Verify mixed valid/invalid metadata - valid entries kept, invalid dropped.""" with langfuse_client.start_as_current_observation(name="parent-span"): From 94cc64bac31e17f2f3d01dd2dc26267cddb58ba3 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 7 Oct 2026 08:45:52 +0000 Subject: [PATCH 16/25] fix(experiments): skip orphaned run scores and offline project lookups Run-level scores are only sent when at least one item trace is sampled, because a fully sampled-out experiment never reaches the server. Clients with tracing disabled no longer resolve the project id. Co-authored-by: Hassieb Pakzad --- langfuse/_client/client.py | 50 +++++++++++++++++++++++++++++------ tests/unit/test_experiment.py | 41 ++++++++++++++++++++++++++++ 2 files changed, 83 insertions(+), 8 deletions(-) diff --git a/langfuse/_client/client.py b/langfuse/_client/client.py index 653243b7d..7c5ac28eb 100644 --- a/langfuse/_client/client.py +++ b/langfuse/_client/client.py @@ -2734,13 +2734,14 @@ async def _run_experiment_async( "Starting experiment '%s' run '%s' with %s items", name, run_name, len(data) ) - try: - project_id = await asyncio.to_thread(self._get_project_id) - except Exception as e: - langfuse_logger.warning( - "Failed to resolve project id for experiment: %s", e - ) - project_id = None + project_id: Optional[str] = None + if self._tracing_enabled: + try: + project_id = await asyncio.to_thread(self._get_project_id) + except Exception as e: + langfuse_logger.warning( + "Failed to resolve project id for experiment: %s", e + ) # One experiment id per run: mixed-dataset data uses the first dataset item's dataset. experiment_dataset_id = next( @@ -2810,8 +2811,24 @@ async def process_item(item: ExperimentItem) -> ExperimentItemResult: else None ) + # Without an exported item span the experiment does not exist on the server, + # so a run score would be orphaned. + stored_run_evaluations = ( + run_evaluations + if any( + result.trace_id is not None and self._is_trace_sampled(result.trace_id) + for result in valid_results + ) + else [] + ) + if run_evaluations and not stored_run_evaluations: + langfuse_logger.debug( + "Skipping run-level scores for experiment %s: all items were sampled out.", + experiment_id, + ) + # Run-level scores attach to the experiment via dataset_run_id == experiment_id. - for evaluation in run_evaluations: + for evaluation in stored_run_evaluations: try: self.create_score( dataset_run_id=experiment_id, @@ -2869,6 +2886,23 @@ def _create_experiment_id( return sha256(payload.encode("utf-8")).hexdigest()[:16] + def _is_trace_sampled(self, trace_id: str) -> bool: + from opentelemetry.sdk.trace.sampling import Decision + + tracer_provider = self._resources.tracer_provider if self._resources else None + sampler = getattr(tracer_provider, "sampler", None) + if sampler is None: + return True + + try: + decision = sampler.should_sample( + parent_context=None, trace_id=int(trace_id, 16), name="experiment" + ).decision + except Exception: + return True + + return bool(decision == Decision.RECORD_AND_SAMPLE) + @staticmethod def _format_experiment_item_version(dataset_version: datetime) -> str: from langfuse.api.core.datetime_utils import serialize_datetime diff --git a/tests/unit/test_experiment.py b/tests/unit/test_experiment.py index 5f279f31d..02596c242 100644 --- a/tests/unit/test_experiment.py +++ b/tests/unit/test_experiment.py @@ -750,3 +750,44 @@ def test_legacy_constructor_arguments_still_populate_new_fields(self): assert item_result.experiment_id == "legacy" assert result.experiment_url == "http://legacy" + + +class TestExperimentRunScoreGating: + def test_run_scores_skipped_when_all_items_are_sampled_out( + self, langfuse_memory_client, monkeypatch + ): + from opentelemetry.sdk.trace.sampling import ALWAYS_OFF + + create_score = MagicMock() + monkeypatch.setattr(langfuse_memory_client, "create_score", create_score) + monkeypatch.setattr(langfuse_memory_client, "_get_project_id", lambda: "p") + monkeypatch.setattr( + langfuse_memory_client._resources.tracer_provider, "sampler", ALWAYS_OFF + ) + monkeypatch.setattr(langfuse_memory_client._otel_tracer, "sampler", ALWAYS_OFF) + + result = langfuse_memory_client.run_experiment( + name="exp", + data=[{"input": "a"}, {"input": "b"}], + task=lambda *, item, **kwargs: item["input"], + run_evaluators=[lambda **kwargs: Evaluation(name="run", value=1.0)], + ) + + assert [e.name for e in result.run_evaluations] == ["run"] + assert _run_scores(create_score) == [] + + def test_tracing_disabled_skips_project_id_lookup(self, monkeypatch): + client = Langfuse( + public_key="pk", secret_key="sk", base_url="http://x", tracing_enabled=False + ) + get_project_id = MagicMock(return_value="p") + monkeypatch.setattr(client, "_get_project_id", get_project_id) + + result = client.run_experiment( + name="exp", + data=[{"input": "a"}], + task=lambda *, item, **kwargs: item["input"], + ) + + get_project_id.assert_not_called() + assert result.experiment_url is None From 41f64799c3a67c7932654278556326001043affb Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 7 Oct 2026 08:49:30 +0000 Subject: [PATCH 17/25] fix(batch-evaluation): one root per trace and safe resume - root_observations scope evaluates each trace once, preferring the physical root over SDK-marked app roots - resuming reuses the token's filter and rejects a different one, since the cursor only continues the query that produced it - the cursorless start-time fallback is inclusive so tied observations are not skipped Co-authored-by: Hassieb Pakzad --- langfuse/batch_evaluation.py | 73 +++++++++++++++++++++++-- tests/unit/test_batch_evaluation.py | 83 ++++++++++++++++++++++++++--- 2 files changed, 144 insertions(+), 12 deletions(-) diff --git a/langfuse/batch_evaluation.py b/langfuse/batch_evaluation.py index 025748de4..3d299e418 100644 --- a/langfuse/batch_evaluation.py +++ b/langfuse/batch_evaluation.py @@ -19,6 +19,7 @@ Literal, Optional, Protocol, + Set, Tuple, Union, get_args, @@ -859,14 +860,15 @@ async def run_async( scope="observations". max_retries: Maximum retries for failed batch fetches. verbose: If True, log progress to console. - resume_from: Resume token from a previous run. + resume_from: Resume token from a previous run. If `filter` is omitted, + the token's filter is reused. Returns: BatchEvaluationResult with comprehensive statistics. Raises: ValueError: If the scope is invalid, the filter is not a JSON array, or - the resume token was created for a different scope. + the resume token was created for a different scope or filter. """ start_time = time.time() @@ -880,6 +882,15 @@ async def run_async( f"Resume token was created for scope {resume_from.scope!r}, " f"cannot resume with scope {scope!r}." ) + # The token's cursor only continues the query that produced it. + if resume_from is not None: + if filter is None: + filter = resume_from.filter + elif filter != resume_from.filter: + raise ValueError( + "Resume token was created for a different filter. Pass the " + "same filter, or omit it to reuse the token's filter." + ) effective_filter = self._build_filter( filter=filter, scope=scope, resume_from=resume_from @@ -911,6 +922,16 @@ async def run_async( resume_from.last_processed_timestamp if resume_from else "" ) last_item_id = resume_from.last_processed_id if resume_from else "" + # The start-time fallback is inclusive so tied items are not lost; skip + # the one item the token says was already processed. + resumed_item_id = ( + resume_from.last_processed_id + if resume_from is not None + and resume_from.cursor is None + and resume_from.last_processed_timestamp + else None + ) + seen_root_trace_ids: Set[str] = set() batch_number = 0 if verbose: @@ -1001,9 +1022,17 @@ async def process_item( except Exception as e: return (item.id, e) - results = await asyncio.gather(*[process_item(item) for item in items]) + items_to_process = [item for item in items if item.id != resumed_item_id] + if scope == "root_observations": + items_to_process = self._select_one_root_per_trace( + items_to_process, seen_root_trace_ids + ) + + results = await asyncio.gather( + *[process_item(item) for item in items_to_process] + ) - for item, (item_id, result) in zip(items, results): + for item, (item_id, result) in zip(items_to_process, results): if isinstance(result, Exception): total_items_failed += 1 failed_item_ids.append(item_id) @@ -1423,13 +1452,47 @@ def _build_filter( { "type": "datetime", "column": "startTime", - "operator": "<", + "operator": "<=", "value": resume_from.last_processed_timestamp, } ) return json.dumps(conditions) if conditions else None + @staticmethod + def _select_one_root_per_trace( + items: List[ObservationV2], seen_trace_ids: Set[str] + ) -> List[ObservationV2]: + """Keep one root observation per trace. + + The v2 API marks both physical roots and SDK-detected app roots as root + observations, so a trace can return several. Prefer the physical root + when it is on the same page; otherwise keep the first one seen. + """ + traces_with_physical_root = { + item.trace_id + for item in items + if item.trace_id and not item.parent_observation_id + } + selected: List[ObservationV2] = [] + + for item in items: + if item.trace_id is None: + selected.append(item) + continue + if item.trace_id in seen_trace_ids: + continue + if ( + item.parent_observation_id + and item.trace_id in traces_with_physical_root + ): + continue + + seen_trace_ids.add(item.trace_id) + selected.append(item) + + return selected + def _build_result( self, total_items_fetched: int, diff --git a/tests/unit/test_batch_evaluation.py b/tests/unit/test_batch_evaluation.py index 2d924de95..8c4240565 100644 --- a/tests/unit/test_batch_evaluation.py +++ b/tests/unit/test_batch_evaluation.py @@ -28,14 +28,20 @@ def make_observation( trace_id: Optional[str] = "default", is_root: bool = False, start_time: Optional[datetime] = None, + observation_id: Optional[str] = None, + parent_observation_id: Optional[str] = "default", ) -> ObservationV2: return ObservationV2( - id=f"obs-{index}", + id=observation_id or f"obs-{index}", trace_id=f"trace-{index}" if trace_id == "default" else trace_id, start_time=start_time or BASE_TIME + timedelta(seconds=index), end_time=None, project_id="project", - parent_observation_id=None if is_root else f"parent-{index}", + parent_observation_id=( + (None if is_root else f"parent-{index}") + if parent_observation_id == "default" + else parent_observation_id + ), type="SPAN", is_root_observation=is_root, input=json.dumps({"question": index}), @@ -72,8 +78,8 @@ def _matches(observation: ObservationV2, filter_json: Optional[str]) -> bool: if observation.is_root_observation is not condition["value"]: return False elif condition["column"] == "startTime": - assert condition["operator"] == "<" - if observation.start_time >= datetime.fromisoformat(condition["value"]): + assert condition["operator"] == "<=" + if observation.start_time > datetime.fromisoformat(condition["value"]): return False return True @@ -292,7 +298,11 @@ async def test_fetch_failure_returns_resume_token_for_failed_page(): @pytest.mark.asyncio async def test_resume_without_cursor_falls_back_to_start_time_bound(): - runner, api, _ = make_runner([make_observation(i) for i in range(5)]) + tied_time = BASE_TIME + timedelta(seconds=3) + runner, api, _ = make_runner( + [make_observation(i) for i in range(5)] + + [make_observation(9, observation_id="obs-tied", start_time=tied_time)] + ) token = BatchEvaluationResumeToken( scope="observations", filter=None, @@ -307,12 +317,71 @@ async def test_resume_without_cursor_falls_back_to_start_time_bound(): { "type": "datetime", "column": "startTime", - "operator": "<", + "operator": "<=", "value": token.last_processed_timestamp, } ] assert api.calls[0]["cursor"] is None - assert set(result.item_evaluations) == {"obs-2", "obs-1", "obs-0"} + assert set(result.item_evaluations) == {"obs-tied", "obs-2", "obs-1", "obs-0"} + + +@pytest.mark.asyncio +async def test_resume_reuses_the_token_filter_and_rejects_a_different_one(): + runner, api, _ = make_runner([make_observation(i) for i in range(4)]) + token_filter = json.dumps( + [{"type": "string", "column": "name", "operator": "=", "value": "x"}] + ) + token = BatchEvaluationResumeToken( + scope="observations", + filter=token_filter, + cursor="2", + last_processed_timestamp=BASE_TIME.isoformat(), + last_processed_id="obs-2", + items_processed=2, + ) + api._matches = lambda observation, filter_json: True # type: ignore[method-assign] + + await run(runner, resume_from=token) + assert json.loads(api.calls[0]["filter"]) == json.loads(token_filter) + + with pytest.raises(ValueError, match="different filter"): + await run(runner, resume_from=token, filter="[]") + + +@pytest.mark.asyncio +async def test_root_observations_scope_prefers_the_physical_root_of_a_trace(): + runner, _, client = make_runner( + [ + make_observation(0, trace_id="t1", is_root=True), + # SDK-marked app root below the physical root of the same trace. + make_observation( + 1, trace_id="t1", is_root=True, parent_observation_id="obs-0" + ), + ] + ) + + result = await run(runner, scope="root_observations") + + assert [c.kwargs["trace_id"] for c in client.create_score.call_args_list] == ["t1"] + assert set(result.item_evaluations) == {"obs-0"} + + +@pytest.mark.asyncio +async def test_root_observations_scope_scores_a_trace_once_across_pages(): + # Sibling app roots under a parent that was not exported, on separate pages. + runner, _, client = make_runner( + [ + make_observation( + i, trace_id="t2", is_root=True, parent_observation_id="hidden" + ) + for i in range(3) + ] + ) + + result = await run(runner, scope="root_observations", fetch_batch_size=1) + + assert [c.kwargs["trace_id"] for c in client.create_score.call_args_list] == ["t2"] + assert set(result.item_evaluations) == {"obs-2"} @pytest.mark.asyncio From 83acb2b47e0d2908d795586c0d7b9b84eb5f08d6 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 7 Oct 2026 09:04:32 +0000 Subject: [PATCH 18/25] chore: re-run CI Co-authored-by: Hassieb Pakzad From 406bc4b2c68fdad52624f27d0dd43ee6a92e1688 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 7 Oct 2026 09:04:34 +0000 Subject: [PATCH 19/25] chore: re-run CI Co-authored-by: Hassieb Pakzad From 5f4b9a426087a26a3aac1f2bef85aa49dc75c73b Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 7 Oct 2026 09:25:31 +0000 Subject: [PATCH 20/25] fix(tracing): match JS for nested big ints and non-finite metadata Integers keep their exact digits as JSON numbers at any depth, and propagated metadata values containing NaN or Infinity are dropped with a warning instead of being stored, matching the JS SDK. Co-authored-by: Hassieb Pakzad --- langfuse/_client/propagation.py | 53 ++++++++++++++++++++----- tests/unit/test_propagate_attributes.py | 53 +++++++++++++++++++++++++ 2 files changed, 95 insertions(+), 11 deletions(-) diff --git a/langfuse/_client/propagation.py b/langfuse/_client/propagation.py index 2eecc8647..20aefe7f1 100644 --- a/langfuse/_client/propagation.py +++ b/langfuse/_client/propagation.py @@ -5,7 +5,7 @@ propagate to all child spans within the context. """ -import json +import math import re from typing import ( Any, @@ -285,8 +285,9 @@ def propagate_attributes( Langfuse's environment format: lowercase alphanumeric with optional hyphens or underscores, must be ≤40 characters, and it must not start with "langfuse". Non-string metadata values are serialized like JavaScript's `JSON.stringify` - (compact separators, non-ASCII kept as is, None becomes "null") - before the 200 character limit is applied. + (compact separators, non-ASCII kept as is, None becomes "null", + integers keep their exact digits) before the 200 character limit is + applied. Values containing NaN or Infinity are dropped. Invalid values will be dropped with a warning logged. - **OpenTelemetry**: This uses OpenTelemetry context propagation under the hood, making it compatible with other OTel-instrumented libraries. @@ -399,6 +400,15 @@ def _propagate_attributes( for key, value in metadata_value.items(): serialized_value = _serialize_propagated_metadata_value(value) + if serialized_value is None: + langfuse_logger.warning( + "Propagated attribute '%s.%s' contains NaN or Infinity, which " + "is not valid JSON. Dropping value.", + metadata_key, + key, + ) + continue + if _validate_string_value( value=serialized_value, key=f"{metadata_key}.{key}" ): @@ -647,18 +657,39 @@ def _validate_propagated_value( return value -def _serialize_propagated_metadata_value(value: Any) -> str: - # Must match JSON.stringify in the JS SDK so both SDKs emit identical values. +class _PropagatedMetadataSerializer(EventSerializer): + """EventSerializer variant that matches the JS SDK for propagated metadata. + + Integers keep their exact digits as JSON numbers at any depth, and values + containing NaN or Infinity are flagged so the caller can drop them. + """ + + def __init__(self, *args: Any, **kwargs: Any) -> None: + super().__init__(*args, **kwargs) + self.found_non_finite_number = False + + def default(self, obj: Any) -> Any: + if isinstance(obj, int) and not isinstance(obj, bool): + return obj + + if isinstance(obj, float) and not math.isfinite(obj): + self.found_non_finite_number = True + return None + + return super().default(obj) + + +def _serialize_propagated_metadata_value(value: Any) -> Optional[str]: + """Serialize like JSON.stringify in the JS SDK; None means drop the value.""" if isinstance(value, str): return value - # EventSerializer quotes ints outside JS's safe range; keep their exact digits. - if isinstance(value, int) and not isinstance(value, bool): - return str(value) - - return json.dumps( - value, cls=EventSerializer, separators=(",", ":"), ensure_ascii=False + serializer = _PropagatedMetadataSerializer( + separators=(",", ":"), ensure_ascii=False ) + serialized = serializer.encode(value) + + return None if serializer.found_non_finite_number else serialized def _validate_string_value(*, value: str, key: str) -> bool: diff --git a/tests/unit/test_propagate_attributes.py b/tests/unit/test_propagate_attributes.py index bd6003e47..2330269f0 100644 --- a/tests/unit/test_propagate_attributes.py +++ b/tests/unit/test_propagate_attributes.py @@ -523,6 +523,59 @@ def test_large_integer_metadata_keeps_its_digits( "9007199254740993", ) + def test_nested_large_integer_metadata_keeps_its_digits( + self, langfuse_client, memory_exporter + ): + """Verify nested integers beyond JS's safe range stay unquoted digits.""" + with langfuse_client.start_as_current_observation(name="parent-span"): + with propagate_attributes( + metadata={ + "ids": [9007199254740993, 1], + "ref": {"snowflake_id": 9007199254740993}, + } + ): + child = langfuse_client.start_observation(name="child-span") + child.end() + + child_span = self.get_span_by_name(memory_exporter, "child-span") + self.verify_span_attribute( + child_span, + f"{LangfuseOtelSpanAttributes.TRACE_METADATA}.ids", + "[9007199254740993,1]", + ) + self.verify_span_attribute( + child_span, + f"{LangfuseOtelSpanAttributes.TRACE_METADATA}.ref", + '{"snowflake_id":9007199254740993}', + ) + + def test_non_finite_number_metadata_is_dropped( + self, langfuse_client, memory_exporter, caplog + ): + """Verify values containing NaN or Infinity are dropped, like in the JS SDK.""" + caplog.set_level("WARNING", logger="langfuse") + with langfuse_client.start_as_current_observation(name="parent-span"): + with propagate_attributes( + metadata={ + "kept": 1.5, + "nan": float("nan"), + "inf": float("-inf"), + "nested": {"scores": [1.0, float("inf")]}, + } + ): + child = langfuse_client.start_observation(name="child-span") + child.end() + + child_span = self.get_span_by_name(memory_exporter, "child-span") + self.verify_span_attribute( + child_span, f"{LangfuseOtelSpanAttributes.TRACE_METADATA}.kept", "1.5" + ) + for key in ("nan", "inf", "nested"): + self.verify_missing_attribute( + child_span, f"{LangfuseOtelSpanAttributes.TRACE_METADATA}.{key}" + ) + assert "metadata.nan" in caplog.text + def test_mixed_valid_invalid_metadata(self, langfuse_client, memory_exporter): """Verify mixed valid/invalid metadata - valid entries kept, invalid dropped.""" with langfuse_client.start_as_current_observation(name="parent-span"): From c7c709a63a452c0ffeb715d99536523b3a7d29fa Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 7 Oct 2026 14:22:06 +0000 Subject: [PATCH 21/25] feat(client): keep host and LANGFUSE_HOST working without a warning Co-authored-by: Hassieb Pakzad --- langfuse/_client/client.py | 9 --------- tests/unit/test_initialization.py | 30 +++++------------------------- 2 files changed, 5 insertions(+), 34 deletions(-) diff --git a/langfuse/_client/client.py b/langfuse/_client/client.py index acf2a0c9e..10b9b52e4 100644 --- a/langfuse/_client/client.py +++ b/langfuse/_client/client.py @@ -329,15 +329,6 @@ def __init__( or host or os.environ.get(LANGFUSE_HOST, "https://cloud.langfuse.com") ) - if ( - not base_url - and not os.environ.get(LANGFUSE_BASE_URL) - and (host or os.environ.get(LANGFUSE_HOST)) - ): - langfuse_logger.warning( - "`host` and LANGFUSE_HOST are deprecated. Use `base_url` or " - "LANGFUSE_BASE_URL instead." - ) self._environment = environment or cast( str, os.environ.get(LANGFUSE_TRACING_ENVIRONMENT) ) diff --git a/tests/unit/test_initialization.py b/tests/unit/test_initialization.py index f564aee47..1f682d3fe 100644 --- a/tests/unit/test_initialization.py +++ b/tests/unit/test_initialization.py @@ -100,35 +100,15 @@ def test_env_host_fallback(self, cleanup_env_vars): assert client._base_url == "http://env-host.com" - def test_env_host_fallback_logs_deprecation_warning(self, cleanup_env_vars, caplog): - """Test that resolving the URL from LANGFUSE_HOST logs a deprecation warning.""" + def test_env_host_fallback_logs_no_warning(self, cleanup_env_vars, caplog): + """Test that LANGFUSE_HOST keeps working without a deprecation warning.""" caplog.set_level("WARNING", logger="langfuse") os.environ["LANGFUSE_HOST"] = "http://env-host.com" - Langfuse(public_key="test_pk", secret_key="test_sk") + client = Langfuse(public_key="test_pk", secret_key="test_sk") - assert any( - "LANGFUSE_HOST are deprecated" in record.message - for record in caplog.records - ) - - def test_host_ignored_when_base_url_set_logs_no_warning( - self, cleanup_env_vars, caplog - ): - """Test that a set base URL suppresses the host deprecation warning.""" - caplog.set_level("WARNING", logger="langfuse") - os.environ["LANGFUSE_HOST"] = "http://env-host.com" - - Langfuse( - base_url="http://param-base-url.com", - public_key="test_pk", - secret_key="test_sk", - ) - - assert not any( - "LANGFUSE_HOST are deprecated" in record.message - for record in caplog.records - ) + assert client._base_url == "http://env-host.com" + assert not any("deprecated" in record.message for record in caplog.records) def test_default_base_url(self, cleanup_env_vars): """Test that default base_url is used when nothing is set.""" From 8063f8467f74e4b23009d30e16aab6da24de25ff Mon Sep 17 00:00:00 2001 From: Hassieb Pakzad <68423100+hassiebp@users.noreply.github.com> Date: Thu, 8 Oct 2026 00:07:00 +0400 Subject: [PATCH 22/25] feat(tracing)!: remove trace-level input/output setters (#1935) Remove the deprecated Langfuse.set_current_trace_io() and span.set_trace_io(), the langfuse.trace.input/output attributes, and their media handling. On Langfuse v4 a trace's input and output come from its root observation: set them with update() on the root span or update_current_span(). Co-authored-by: Cursor Agent Co-authored-by: Hassieb Pakzad --- langfuse/_client/attributes.py | 6 ---- langfuse/_client/client.py | 52 +----------------------------- langfuse/_client/span.py | 48 --------------------------- langfuse/_client/span_exporter.py | 2 -- tests/e2e/test_core_sdk.py | 20 ++++++------ tests/e2e/test_decorators.py | 4 +-- tests/unit/test_mask_otel_spans.py | 2 -- tests/unit/test_otel.py | 8 ++--- 8 files changed, 16 insertions(+), 126 deletions(-) diff --git a/langfuse/_client/attributes.py b/langfuse/_client/attributes.py index 43a85c2fd..e7027a192 100644 --- a/langfuse/_client/attributes.py +++ b/langfuse/_client/attributes.py @@ -32,8 +32,6 @@ class LangfuseOtelSpanAttributes: TRACE_TAGS = "langfuse.trace.tags" TRACE_PUBLIC = "langfuse.trace.public" TRACE_METADATA = "langfuse.trace.metadata" - TRACE_INPUT = "langfuse.trace.input" - TRACE_OUTPUT = "langfuse.trace.output" # Langfuse-observation attributes OBSERVATION_TYPE = "langfuse.observation.type" @@ -76,13 +74,9 @@ class LangfuseOtelSpanAttributes: def create_trace_attributes( *, - input: Optional[Any] = None, - output: Optional[Any] = None, public: Optional[bool] = None, ) -> dict: attributes = { - LangfuseOtelSpanAttributes.TRACE_INPUT: _serialize(input), - LangfuseOtelSpanAttributes.TRACE_OUTPUT: _serialize(output), LangfuseOtelSpanAttributes.TRACE_PUBLIC: public, } diff --git a/langfuse/_client/client.py b/langfuse/_client/client.py index 02d122862..d6191248a 100644 --- a/langfuse/_client/client.py +++ b/langfuse/_client/client.py @@ -38,7 +38,6 @@ _agnosticcontextmanager, ) from packaging.version import Version -from typing_extensions import deprecated from langfuse._client.attributes import ( LangfuseOtelSpanAttributes, @@ -213,7 +212,7 @@ class Langfuse: release (Optional[str]): Release version/hash of your application. Used for grouping analytics by release. media_upload_thread_count (Optional[int]): Number of background threads for handling media uploads. Defaults to 1. Can also be set via LANGFUSE_MEDIA_UPLOAD_THREAD_COUNT environment variable. sample_rate (Optional[float]): Sampling rate for traces (0.0 to 1.0). Defaults to 1.0 (100% of traces are sampled). Can also be set via LANGFUSE_SAMPLE_RATE environment variable. - mask (Optional[MaskFunction]): Function to mask sensitive data synchronously when Langfuse SDK attributes are created. This applies only to data set through Langfuse SDK APIs such as `start_observation()`, `update()`, and `set_trace_io()`. + mask (Optional[MaskFunction]): Function to mask sensitive data synchronously when Langfuse SDK attributes are created. This applies only to data set through Langfuse SDK APIs such as `start_observation()` and `update()`. mask_otel_spans (Optional[MaskOtelSpansFunction]): Synchronous export-stage hook for masking raw OpenTelemetry span attributes before this Langfuse client sends them to Langfuse. Use this for spans created by third-party OpenTelemetry instrumentations, or when you need to inspect final span attributes after export filtering and Langfuse media handling. It does not modify spans already exported through other OpenTelemetry exporters. The hook receives one OpenTelemetry export batch. A batch is not guaranteed to contain a complete trace, request, or Langfuse observation tree. The hook usually runs on the OpenTelemetry batch span processor worker thread; during `flush()` and shutdown it may run on the caller thread. Keep it synchronous, deterministic, and fast. @@ -1534,55 +1533,6 @@ def update_current_span( status_message=status_message, ) - @deprecated( - "Trace-level input/output is deprecated. " - "For trace attributes (user_id, session_id, tags, etc.), use propagate_attributes() instead. " - "This method will be removed in a future major version." - ) - def set_current_trace_io( - self, - *, - input: Optional[Any] = None, - output: Optional[Any] = None, - ) -> None: - """Set trace-level input and output for the current span's trace. - - .. deprecated:: - This is a legacy method for backward compatibility with Langfuse platform - features that still rely on trace-level input/output (e.g., legacy LLM-as-a-judge - evaluators). It will be removed in a future major version. - - For setting other trace attributes (user_id, session_id, metadata, tags, version), - use :func:`langfuse.propagate_attributes` (top-level import) instead. - - Args: - input: Input data to associate with the trace. - output: Output data to associate with the trace. - """ - if not self._tracing_enabled: - langfuse_logger.debug( - "Operation skipped: set_current_trace_io - Tracing is disabled or client is in no-op mode." - ) - return - - current_otel_span = self._get_current_otel_span() - - if current_otel_span is not None and current_otel_span.is_recording(): - span_class = self._get_span_class( - self._get_observation_type_from_otel_span(current_otel_span) - ) - span = span_class( - otel_span=current_otel_span, - langfuse_client=self, - environment=self._environment, - release=self._release, - ) - - span.set_trace_io( - input=input, - output=output, - ) - def set_current_trace_as_public(self) -> None: """Make the current trace publicly accessible via its URL. diff --git a/langfuse/_client/span.py b/langfuse/_client/span.py index 96879499c..9e2b5b718 100644 --- a/langfuse/_client/span.py +++ b/langfuse/_client/span.py @@ -36,7 +36,6 @@ if TYPE_CHECKING: from langfuse._client.client import Langfuse -from typing_extensions import deprecated from langfuse._client.attributes import ( LangfuseOtelSpanAttributes, @@ -226,53 +225,6 @@ def end(self, *, end_time: Optional[int] = None) -> "LangfuseObservationWrapper" return self - @deprecated( - "Trace-level input/output is deprecated. " - "For trace attributes (user_id, session_id, tags, etc.), use propagate_attributes() instead. " - "This method will be removed in a future major version." - ) - def set_trace_io( - self, - *, - input: Optional[Any] = None, - output: Optional[Any] = None, - ) -> "LangfuseObservationWrapper": - """Set trace-level input and output for the trace this span belongs to. - - .. deprecated:: - This is a legacy method for backward compatibility with Langfuse platform - features that still rely on trace-level input/output (e.g., legacy LLM-as-a-judge - evaluators). It will be removed in a future major version. - - For setting other trace attributes (user_id, session_id, metadata, tags, version), - use :func:`langfuse.propagate_attributes` (top-level import) instead. - - Args: - input: Input data to associate with the trace. - output: Output data to associate with the trace. - - Returns: - The span instance for method chaining. - """ - if not self._otel_span.is_recording(): - return self - - media_processed_input = self._process_media_and_apply_mask( - data=input, field="input", span=self._otel_span - ) - media_processed_output = self._process_media_and_apply_mask( - data=output, field="output", span=self._otel_span - ) - - attributes = create_trace_attributes( - input=media_processed_input, - output=media_processed_output, - ) - - self._otel_span.set_attributes(attributes) - - return self - def set_trace_as_public(self) -> "LangfuseObservationWrapper": """Make this trace publicly accessible via its URL. diff --git a/langfuse/_client/span_exporter.py b/langfuse/_client/span_exporter.py index 661a1a26c..efbb04261 100644 --- a/langfuse/_client/span_exporter.py +++ b/langfuse/_client/span_exporter.py @@ -27,7 +27,6 @@ _INPUT_MEDIA_ATTRIBUTE_KEYS = frozenset( { - LangfuseOtelSpanAttributes.TRACE_INPUT, LangfuseOtelSpanAttributes.OBSERVATION_INPUT, "ai.prompt.messages", "ai.prompt", @@ -54,7 +53,6 @@ _OUTPUT_MEDIA_ATTRIBUTE_KEYS = frozenset( { - LangfuseOtelSpanAttributes.TRACE_OUTPUT, LangfuseOtelSpanAttributes.OBSERVATION_OUTPUT, "ai.response.text", "ai.result.text", diff --git a/tests/e2e/test_core_sdk.py b/tests/e2e/test_core_sdk.py index 614d6da41..e6f66f20d 100644 --- a/tests/e2e/test_core_sdk.py +++ b/tests/e2e/test_core_sdk.py @@ -549,14 +549,14 @@ def test_create_update_current_trace(): trace_name = create_uuid() - # Create initial span with trace properties using propagate_attributes and set_current_trace_io + # Create initial span with trace properties using propagate_attributes with langfuse.start_as_current_observation(name="test-span-current") as span: with propagate_attributes( trace_name=trace_name, user_id="test", metadata={"key": "value"}, ): - langfuse.set_current_trace_io(input="test_input") + langfuse.update_current_span(input="test_input") langfuse.set_current_trace_as_public() # Get trace ID for later reference trace_id = span.trace_id @@ -989,7 +989,7 @@ def test_create_trace_and_generation(): # Create parent span and set trace properties with langfuse.start_as_current_observation(name=trace_name) as parent_span: with propagate_attributes(trace_name=trace_name, session_id="test-session-id"): - parent_span.set_trace_io(input={"key": "value"}) + parent_span.update(input={"key": "value"}) # Create a generation as child generation = parent_span.start_observation( @@ -1812,7 +1812,7 @@ def test_fetch_traces(): # First trace with langfuse.start_as_current_observation(name="test1") as span: with propagate_attributes(trace_name=name, session_id="session-1"): - span.set_trace_io(input={"key": "value"}, output="output-value") + span.update(input={"key": "value"}, output="output-value") trace_ids.append(span.trace_id) sleep(1) # Ensure traces have different timestamps @@ -1820,7 +1820,7 @@ def test_fetch_traces(): # Second trace with langfuse.start_as_current_observation(name="test2") as span: with propagate_attributes(trace_name=name, session_id="session-1"): - span.set_trace_io(input={"key": "value"}, output="output-value") + span.update(input={"key": "value"}, output="output-value") trace_ids.append(span.trace_id) sleep(1) # Ensure traces have different timestamps @@ -1828,7 +1828,7 @@ def test_fetch_traces(): # Third trace with langfuse.start_as_current_observation(name="test3") as span: with propagate_attributes(trace_name=name, session_id="session-1"): - span.set_trace_io(input={"key": "value"}, output="output-value") + span.update(input={"key": "value"}, output="output-value") trace_ids.append(span.trace_id) # Ensure data is sent @@ -2088,11 +2088,11 @@ def mask_func(data): # Create a root span with trace properties with langfuse.start_as_current_observation(name="test-span") as root_span: with propagate_attributes(trace_name="test_trace"): - root_span.set_trace_io(input={"sensitive": "data"}) + root_span.update(input={"sensitive": "data"}) # Get trace ID for later use trace_id = root_span.trace_id # Add output to the trace - root_span.set_trace_io(output={"more": "sensitive"}) + root_span.update(output={"more": "sensitive"}) # Create a generation as child gen = root_span.start_observation( @@ -2136,11 +2136,11 @@ def mask_func(data): # Create a root span with trace properties with langfuse.start_as_current_observation(name="test-span") as root_span: with propagate_attributes(trace_name="test_trace"): - root_span.set_trace_io(input={"should_raise": "data"}) + root_span.update(input={"should_raise": "data"}) # Get trace ID for later use trace_id = root_span.trace_id # Add output to the trace - root_span.set_trace_io(output={"should_raise": "sensitive"}) + root_span.update(output={"should_raise": "sensitive"}) # Ensure data is sent langfuse.flush() diff --git a/tests/e2e/test_decorators.py b/tests/e2e/test_decorators.py index 9fac4cf22..acd495f57 100644 --- a/tests/e2e/test_decorators.py +++ b/tests/e2e/test_decorators.py @@ -1071,10 +1071,10 @@ def test_media(): media = LangfuseMedia(content_bytes=pdf_bytes, content_type="application/pdf") - @observe() + @observe(capture_input=False, capture_output=False) def main(): sleep(1) - langfuse.set_current_trace_io( + langfuse.update_current_span( input={ "context": { "nested": media, diff --git a/tests/unit/test_mask_otel_spans.py b/tests/unit/test_mask_otel_spans.py index 5d1d52ee4..d9a23befb 100644 --- a/tests/unit/test_mask_otel_spans.py +++ b/tests/unit/test_mask_otel_spans.py @@ -293,7 +293,6 @@ def test_export_stage_media_processes_string_sequence_attributes(): @pytest.mark.parametrize( ("attribute_key", "expected_field"), [ - ("langfuse.trace.input", "input"), ("langfuse.observation.input", "input"), ("ai.prompt.messages", "input"), ("gcp.vertex.agent.tool_call_args", "input"), @@ -302,7 +301,6 @@ def test_export_stage_media_processes_string_sequence_attributes(): ("gen_ai.input.messages", "input"), ("gen_ai.prompt.0.content", "input"), ("llm.input_messages.0.message.content", "input"), - ("langfuse.trace.output", "output"), ("langfuse.observation.output", "output"), ("ai.response.toolCalls", "output"), ("gcp.vertex.agent.tool_response", "output"), diff --git a/tests/unit/test_otel.py b/tests/unit/test_otel.py index ccdec39fd..6c4b23d3e 100644 --- a/tests/unit/test_otel.py +++ b/tests/unit/test_otel.py @@ -498,8 +498,8 @@ def test_generation_name_update(self, langfuse_client, memory_exporter): def test_trace_update(self, langfuse_client, memory_exporter): """Test updating trace level attributes.""" - # Create a span and set trace attributes using propagate_attributes and set_trace_io - with langfuse_client.start_as_current_observation(name="trace-span") as span: + # Create a span and set trace attributes using propagate_attributes + with langfuse_client.start_as_current_observation(name="trace-span"): with propagate_attributes( trace_name="updated-trace-name", user_id="test-user", @@ -507,7 +507,7 @@ def test_trace_update(self, langfuse_client, memory_exporter): tags=["tag1", "tag2"], metadata={"trace-meta": "data"}, ): - span.set_trace_io(input={"trace-input": "value"}) + pass # Get the span data spans = self.get_spans_by_name(memory_exporter, "trace-span") @@ -526,12 +526,10 @@ def test_trace_update(self, langfuse_client, memory_exporter): else: tags = list(attributes[LangfuseOtelSpanAttributes.TRACE_TAGS]) - input_data = json.loads(attributes[LangfuseOtelSpanAttributes.TRACE_INPUT]) metadata = attributes[f"{LangfuseOtelSpanAttributes.TRACE_METADATA}.trace-meta"] # Check attribute values assert sorted(tags) == sorted(["tag1", "tag2"]) - assert input_data == {"trace-input": "value"} assert metadata == "data" def test_complex_scenario(self, langfuse_client, memory_exporter): From 4ea561df36f86c9cfce98ad805f13ac2860cc2da Mon Sep 17 00:00:00 2001 From: Hassieb Pakzad <68423100+hassiebp@users.noreply.github.com> Date: Thu, 8 Oct 2026 00:12:29 +0400 Subject: [PATCH 23/25] test: run the e2e suites on Langfuse v4 events_only (#1937) * feat(tracing)!: remove trace-level input/output setters Remove the deprecated Langfuse.set_current_trace_io() and span.set_trace_io(), the langfuse.trace.input/output attributes, and their media handling. On Langfuse v4 a trace's input and output come from its root observation: set them with update() on the root span or update_current_span(). Co-authored-by: Hassieb Pakzad * test: read e2e assertions through the v4 observations API Co-authored-by: Hassieb Pakzad * ci: run e2e against an events_only server Co-authored-by: Hassieb Pakzad * test: pin that create_score logs invalid input without enqueueing Co-authored-by: Hassieb Pakzad * test: remove the v3 wait_for_trace helper Co-authored-by: Hassieb Pakzad * test: keep scalar text IO as strings and read full response_format The v2 observations API returns IO as raw strings, so only JSON objects and arrays are parsed back; text like "2" stays a string. The response_format assertion requests the untruncated metadata value. Co-authored-by: Hassieb Pakzad * test: dedupe observation rows and assert streamed text as a string The events table can briefly return several rows for one span, so the observation helper keeps the latest row per id. A streamed completion of "2" is text and is now asserted as a string. Co-authored-by: Hassieb Pakzad --------- Co-authored-by: Cursor Agent Co-authored-by: Hassieb Pakzad --- .github/langfuse-v4-dual-write.override.yml | 7 - .github/workflows/ci.yml | 2 - tests/e2e/test_core_sdk.py | 832 +++++++---------- tests/e2e/test_decorators.py | 115 +-- tests/e2e/test_media.py | 31 +- tests/e2e/test_prompt.py | 26 +- tests/live_provider/test_langchain.py | 66 +- .../test_langchain_integration.py | 236 +++-- tests/live_provider/test_openai.py | 851 ++++++++---------- tests/support/api_wrapper.py | 130 --- tests/support/utils.py | 345 ++++++- tests/unit/test_e2e_support.py | 320 +++++-- tests/unit/test_resource_manager.py | 31 + 13 files changed, 1549 insertions(+), 1443 deletions(-) delete mode 100644 .github/langfuse-v4-dual-write.override.yml delete mode 100644 tests/support/api_wrapper.py diff --git a/.github/langfuse-v4-dual-write.override.yml b/.github/langfuse-v4-dual-write.override.yml deleted file mode 100644 index 48ab6ff48..000000000 --- a/.github/langfuse-v4-dual-write.override.yml +++ /dev/null @@ -1,7 +0,0 @@ -services: - langfuse-worker: - environment: - LANGFUSE_MIGRATION_V4_WRITE_MODE: dual - langfuse-web: - environment: - LANGFUSE_MIGRATION_V4_WRITE_MODE: dual diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 104cda66f..ee355e49b 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -159,8 +159,6 @@ jobs: LANGFUSE_SERVER_SHA="$(git ls-remote https://github.com/langfuse/langfuse.git HEAD | cut -f1)" curl -fsSL "https://raw.githubusercontent.com/langfuse/langfuse/${LANGFUSE_SERVER_SHA}/docker-compose.yml" \ -o ./langfuse-server/docker-compose.yml - cp ./.github/langfuse-v4-dual-write.override.yml \ - ./langfuse-server/docker-compose.override.yml echo "${LANGFUSE_SERVER_SHA}" - name: Run langfuse server diff --git a/tests/e2e/test_core_sdk.py b/tests/e2e/test_core_sdk.py index e6f66f20d..5330932c3 100644 --- a/tests/e2e/test_core_sdk.py +++ b/tests/e2e/test_core_sdk.py @@ -1,3 +1,4 @@ +import json import os import time from asyncio import gather @@ -5,17 +6,20 @@ from time import sleep import pytest -from tenacity import Retrying, stop_after_delay, wait_fixed from langfuse import Langfuse, propagate_attributes from langfuse._client.resource_manager import LangfuseResourceManager from langfuse._utils import _get_timestamp -from tests.support.api_wrapper import LangfuseAPI from tests.support.utils import ( create_uuid, get_api, - wait_for_result, - wait_for_trace, + get_observations, + get_root_observation, + get_scores, + user_metadata, + wait_for_observations, + wait_for_root_observation, + wait_for_scores, ) @@ -38,31 +42,28 @@ async def update_generation(i, langfuse: Langfuse): # End the generation generation.end() + return generation.trace_id + # Create Langfuse client langfuse = Langfuse() # Run concurrent operations - await gather(*(update_generation(i, langfuse) for i in range(100))) + trace_ids = await gather(*(update_generation(i, langfuse) for i in range(100))) langfuse.flush() - # Allow time for all operations to be processed - sleep(10) - # Verify that all spans were created properly - api = get_api() - for i in range(100): - # Find the observations with the expected name - observations = api.legacy.observations_v1.get_many(name=str(i)).data + for i, trace_id in enumerate(trace_ids): + observations = wait_for_observations(trace_id, min_count=2) - # Find generation observations (there should be at least one) generation_obs = [obs for obs in observations if obs.type == "GENERATION"] - assert len(generation_obs) > 0 + assert len(generation_obs) == 1 # Verify metadata observation = generation_obs[0] assert observation.name == str(i) - assert observation.metadata["count"] == i + assert user_metadata(observation)["count"] == i + assert get_root_observation(observations).trace_name == str(i) def test_flush(): @@ -80,14 +81,10 @@ def test_flush(): # Flush all pending spans to the Langfuse API langfuse.flush() - # Allow time for API to process - sleep(2) - # Verify traces were sent by checking they exist in the API - api = get_api() for i, trace_id in enumerate(trace_ids): - trace = api.trace.get(trace_id) - assert trace.name == str(i) + root = wait_for_root_observation(trace_id) + assert root.trace_name == str(i) def test_invalid_score_data_does_not_raise_exception(): @@ -152,21 +149,20 @@ def test_create_session_score(): # Ensure data is sent langfuse.flush() - sleep(2) # Retrieve and verify - score = get_api().scores.get_by_id(score_id) + scores = wait_for_scores(session_id=session_id, id=score_id) - # find the score by name (server may transform the id format) - assert score is not None + assert len(scores) == 1 + score = scores[0] assert score.value == 1 assert score.data_type == "NUMERIC" - assert score.session_id == session_id + assert score.subject.kind == "session" + assert score.subject.id == session_id def test_create_numeric_score(): langfuse = Langfuse() - api_wrapper = LangfuseAPI() # Create a span and set trace properties with langfuse.start_as_current_observation(name="test-span") as span: @@ -202,22 +198,21 @@ def test_create_numeric_score(): # Ensure data is sent langfuse.flush() - sleep(2) # Retrieve and verify - trace = api_wrapper.get_trace(trace_id) + wait_for_observations(trace_id, min_count=2) + scores = wait_for_scores(trace_id=trace_id, name="this-is-a-score") - # Find the score by name (server may transform the ID format) - score = next((s for s in trace["scores"] if s["name"] == "this-is-a-score"), None) - assert score is not None - assert score["value"] == 1 - assert score["dataType"] == "NUMERIC" - assert score["stringValue"] is None + assert len(scores) == 1 + score = scores[0] + assert score.id == score_id + assert score.value == 1 + assert score.data_type == "NUMERIC" + assert score.subject.kind == "trace" def test_create_boolean_score(): langfuse = Langfuse() - api_wrapper = LangfuseAPI() # Create a span and set trace properties with langfuse.start_as_current_observation(name="test-span") as span: @@ -231,7 +226,7 @@ def test_create_boolean_score(): # Ensure data is sent langfuse.flush() - api_wrapper.get_trace(trace_id) + wait_for_root_observation(trace_id) # Create a boolean score score_id = create_uuid() @@ -256,27 +251,17 @@ def test_create_boolean_score(): langfuse.flush() # Retrieve and verify - trace = api_wrapper.get_trace( - trace_id, - is_result_ready=lambda trace: any( - score["name"] == "this-is-a-score" for score in trace.get("scores", []) - ), - ) + scores = wait_for_scores(trace_id=trace_id, name="this-is-a-score") - # Find the score we created by name - created_score = next( - (s for s in trace["scores"] if s["name"] == "this-is-a-score"), None - ) - assert created_score is not None, "Score not found in trace" - assert created_score["id"] == score_id - assert created_score["dataType"] == "BOOLEAN" - assert created_score["value"] == 1 - assert created_score["stringValue"] == "True" + assert len(scores) == 1, "Score not found in trace" + created_score = scores[0] + assert created_score.id == score_id + assert created_score.data_type == "BOOLEAN" + assert created_score.value is True def test_create_categorical_score(): langfuse = Langfuse() - api_wrapper = LangfuseAPI() # Create a span and set trace properties with langfuse.start_as_current_observation(name="test-span") as span: @@ -290,7 +275,7 @@ def test_create_categorical_score(): # Ensure data is sent langfuse.flush() - api_wrapper.get_trace(trace_id) + wait_for_root_observation(trace_id) # Create a categorical score score_id = create_uuid() @@ -314,27 +299,17 @@ def test_create_categorical_score(): langfuse.flush() # Retrieve and verify - trace = api_wrapper.get_trace( - trace_id, - is_result_ready=lambda trace: any( - score["name"] == "this-is-a-score" for score in trace.get("scores", []) - ), - ) + scores = wait_for_scores(trace_id=trace_id, name="this-is-a-score") - # Find the score we created by name - created_score = next( - (s for s in trace["scores"] if s["name"] == "this-is-a-score"), None - ) - assert created_score is not None, "Score not found in trace" - assert created_score["id"] == score_id - assert created_score["dataType"] == "CATEGORICAL" - assert created_score["value"] == 0 - assert created_score["stringValue"] == "high score" + assert len(scores) == 1, "Score not found in trace" + created_score = scores[0] + assert created_score.id == score_id + assert created_score.data_type == "CATEGORICAL" + assert created_score.value == "high score" def test_create_text_score(): langfuse = Langfuse() - api_wrapper = LangfuseAPI() # Create a span and set trace properties with langfuse.start_as_current_observation(name="test-span") as span: @@ -372,30 +347,21 @@ def test_create_text_score(): # Ensure data is sent langfuse.flush() - # Retrieve and verify with retry - for attempt in Retrying( - stop=stop_after_delay(10), wait=wait_fixed(0.1), reraise=True - ): - with attempt: - trace = api_wrapper.get_trace(trace_id) - - # Find the score we created by name - created_score = next( - (s for s in trace["scores"] if s["name"] == "this-is-a-score"), None - ) - assert created_score is not None, "Score not found in trace" - assert created_score["id"] == score_id - assert created_score["dataType"] == "TEXT" + # Retrieve and verify + scores = wait_for_scores(trace_id=trace_id, name="this-is-a-score") - assert ( - created_score["stringValue"] - == "This is a detailed text evaluation of the output quality." - ) + assert len(scores) == 1, "Score not found in trace" + created_score = scores[0] + assert created_score.id == score_id + assert created_score.data_type == "TEXT" + assert ( + created_score.value + == "This is a detailed text evaluation of the output quality." + ) def test_create_score_with_custom_timestamp(): langfuse = Langfuse() - api_wrapper = LangfuseAPI() # Create a span and set trace properties with langfuse.start_as_current_observation(name="test-span") as span: @@ -409,7 +375,7 @@ def test_create_score_with_custom_timestamp(): # Ensure data is sent langfuse.flush() - api_wrapper.get_trace(trace_id) + wait_for_root_observation(trace_id) custom_timestamp = datetime.now(timezone.utc) - timedelta(hours=1) score_id = create_uuid() @@ -426,28 +392,16 @@ def test_create_score_with_custom_timestamp(): langfuse.flush() # Retrieve and verify - trace = api_wrapper.get_trace( - trace_id, - is_result_ready=lambda trace: any( - score["name"] == "custom-timestamp-score" - for score in trace.get("scores", []) - ), - ) + scores = wait_for_scores(trace_id=trace_id, name="custom-timestamp-score") - # Find the score we created by name - created_score = next( - (s for s in trace["scores"] if s["name"] == "custom-timestamp-score"), None - ) - assert created_score is not None, "Score not found in trace" - assert created_score["id"] == score_id - assert created_score["dataType"] == "NUMERIC" - assert created_score["value"] == 0.85 + assert len(scores) == 1, "Score not found in trace" + created_score = scores[0] + assert created_score.id == score_id + assert created_score.data_type == "NUMERIC" + assert created_score.value == 0.85 # Verify timestamp is close to our custom timestamp - # Parse the timestamp from the API response - response_timestamp = datetime.fromisoformat( - created_score["timestamp"].replace("Z", "+00:00") - ) + response_timestamp = created_score.timestamp # Check that the timestamps are within 1 second of each other # (allowing for some processing time and rounding) @@ -477,24 +431,14 @@ def test_create_trace(): langfuse.flush() # Retrieve the trace from the API - trace = LangfuseAPI().get_trace( - trace_id, - is_result_ready=lambda trace: ( - trace.get("name") == trace_name - and trace.get("userId") == "test" - and trace.get("metadata", {}).get("key") == "value" - and trace.get("tags") == ["tag1", "tag2"] - and trace.get("public") is True - ), - ) + root = wait_for_root_observation(trace_id) # Verify all trace properties - assert trace["name"] == trace_name - assert trace["userId"] == "test" - assert trace["metadata"]["key"] == "value" - assert trace["tags"] == ["tag1", "tag2"] - assert trace["public"] is True - assert True if not trace["externalId"] else False + assert root.trace_name == trace_name + assert root.user_id == "test" + assert user_metadata(root) == {"key": "value"} + assert root.tags == ["tag1", "tag2"] + assert root.public is True def test_create_update_trace(): @@ -525,23 +469,12 @@ def test_create_update_trace(): assert isinstance(trace_id, str) # Retrieve and verify trace - trace = wait_for_trace( - trace_id, - is_result_ready=lambda trace: ( - trace.name == trace_name - and trace.user_id == "test" - and trace.metadata is not None - and trace.metadata.get("key") == "value" - and trace.metadata.get("key2") == "value2" - and trace.public is True - ), - ) + root = wait_for_root_observation(trace_id) - assert trace.name == trace_name - assert trace.user_id == "test" - assert trace.metadata["key"] == "value" - assert trace.metadata["key2"] == "value2" - assert trace.public is True + assert root.trace_name == trace_name + assert root.user_id == "test" + assert user_metadata(root) == {"key": "value", "key2": "value2"} + assert root.public is True def test_create_update_current_trace(): @@ -570,20 +503,18 @@ def test_create_update_current_trace(): # Ensure data is sent to the API langfuse.flush() - sleep(2) assert isinstance(trace_id, str) # Retrieve and verify trace - trace = get_api().trace.get(trace_id) + root = wait_for_root_observation(trace_id) # The 2nd update to the trace must not erase previously set attributes - assert trace.name == trace_name - assert trace.user_id == "test" - assert trace.metadata["key"] == "value" - assert trace.metadata["key2"] == "value2" - assert trace.public is True - assert trace.version == "1.0" - assert trace.input == "test_input" + assert root.trace_name == trace_name + assert root.user_id == "test" + assert user_metadata(root) == {"key": "value", "key2": "value2"} + assert root.public is True + assert root.version == "1.0" + assert root.input == "test_input" def test_create_generation(): @@ -620,19 +551,19 @@ def test_create_generation(): # Flush to ensure all data is sent langfuse.flush() - sleep(2) # Retrieve the trace from the API - trace = get_api().trace.get(trace_id) - - # Verify trace details - assert trace.name == "query-generation" - assert trace.user_id is None + observations = wait_for_observations(trace_id) - assert len(trace.observations) == 1 + assert len(observations) == 1 # Verify generation details - generation_api = trace.observations[0] + generation_api = observations[0] + + # Verify trace details + assert generation_api.is_root_observation is True + assert generation_api.trace_name == "query-generation" + assert generation_api.user_id is None assert generation_api.name == "query-generation" assert generation_api.start_time is not None @@ -707,15 +638,14 @@ def test_create_generation_complex( langfuse.flush() trace_id = generation.trace_id - trace = get_api().trace.get(trace_id) + observations = wait_for_observations(trace_id) - assert trace.name == "query-generation" - assert trace.user_id is None + assert len(observations) == 1 - assert len(trace.observations) == 1 - - generation_api = trace.observations[0] + generation_api = observations[0] + assert generation_api.trace_name == "query-generation" + assert generation_api.user_id is None assert generation_api.id == generation.id assert generation_api.name == "query-generation" assert generation_api.input == [ @@ -727,13 +657,7 @@ def test_create_generation_complex( ] assert generation_api.output == [{"foo": "bar"}] - # Check if metadata exists and has tags before asserting - if ( - hasattr(generation_api, "metadata") - and generation_api.metadata is not None - and "tags" in generation_api.metadata - ): - assert generation_api.metadata["tags"] == ["yo"] + assert user_metadata(generation_api) == {"tags": ["yo"]} assert generation_api.start_time is not None assert generation_api.usage_details == {"input": 51, "output": 0, "total": 100} @@ -759,19 +683,18 @@ def test_create_span(): # Ensure all data is sent langfuse.flush() - sleep(2) # Retrieve from API - trace = get_api().trace.get(trace_id) - - # Verify trace details - assert trace.name == "span" - assert trace.user_id is None + observations = wait_for_observations(trace_id) - assert len(trace.observations) == 1 + assert len(observations) == 1 # Verify span details - span_api = trace.observations[0] + span_api = observations[0] + + # Verify trace details + assert span_api.trace_name == "span" + assert span_api.user_id is None assert span_api.id == span_id assert span_api.name == "span" @@ -784,7 +707,6 @@ def test_create_span(): def test_score_trace(): langfuse = Langfuse() - api_wrapper = LangfuseAPI() trace_name = create_uuid() @@ -803,20 +725,17 @@ def test_score_trace(): # Ensure data is sent langfuse.flush() - sleep(2) # Retrieve and verify - trace = api_wrapper.get_trace(trace_id) - - assert trace["name"] == trace_name + assert wait_for_root_observation(trace_id).trace_name == trace_name - # Find the score we created by name (server may create additional auto-scores) - score = next((s for s in trace["scores"] if s["name"] == "valuation"), None) - assert score is not None - assert score["value"] == 0.5 - assert score["comment"] == "This is a comment" - assert score["observationId"] is None - assert score["dataType"] == "NUMERIC" + scores = wait_for_scores(trace_id=trace_id, name="valuation") + assert len(scores) == 1 + score = scores[0] + assert score.value == 0.5 + assert score.comment == "This is a comment" + assert score.subject.kind == "trace" + assert score.data_type == "NUMERIC" def test_score_trace_nested_trace(): @@ -839,19 +758,16 @@ def test_score_trace_nested_trace(): # Ensure data is sent langfuse.flush() - sleep(2) # Retrieve and verify - trace = get_api().trace.get(trace_id) + assert wait_for_root_observation(trace_id).trace_name == trace_name - assert trace.name == trace_name - - # Find the score we created by name (server may create additional auto-scores) - score = next((s for s in trace.scores if s.name == "valuation"), None) - assert score is not None + scores = wait_for_scores(trace_id=trace_id, name="valuation") + assert len(scores) == 1 + score = scores[0] assert score.value == 0.5 assert score.comment == "This is a comment" - assert score.observation_id is None # API returns this field name + assert score.subject.kind == "trace" assert score.data_type == "NUMERIC" @@ -882,25 +798,23 @@ def test_score_trace_nested_observation(): # Ensure data is sent langfuse.flush() - sleep(2) # Retrieve and verify - trace = get_api().trace.get(trace_id) - - assert trace.name == trace_name + assert wait_for_root_observation(trace_id).trace_name == trace_name - # Find the score we created by name (server may create additional auto-scores) - score = next((s for s in trace.scores if s.name == "valuation"), None) - assert score is not None + scores = wait_for_scores(trace_id=trace_id, name="valuation") + assert len(scores) == 1 + score = scores[0] assert score.value == 0.5 assert score.comment == "This is a comment" - assert score.observation_id == child_span_id # API returns this field name + assert score.subject.kind == "observation" + assert score.subject.id == child_span_id + assert score.subject.trace_id == trace_id assert score.data_type == "NUMERIC" def test_score_span(): langfuse = Langfuse() - api_wrapper = LangfuseAPI() # Create a span span = langfuse.start_observation( @@ -928,20 +842,19 @@ def test_score_span(): # Ensure data is sent langfuse.flush() - sleep(3) # Retrieve and verify - trace = api_wrapper.get_trace(trace_id) + assert len(wait_for_observations(trace_id)) == 1 - assert len(trace["observations"]) == 1 - - # Find the score we created by name (server may create additional auto-scores) - score = next((s for s in trace["scores"] if s["name"] == "valuation"), None) - assert score is not None - assert score["value"] == 1 - assert score["comment"] == "This is a comment" - assert score["observationId"] == span_id - assert score["dataType"] == "NUMERIC" + scores = wait_for_scores(trace_id=trace_id, observation_id=span_id) + assert len(scores) == 1 + score = scores[0] + assert score.name == "valuation" + assert score.value == 1 + assert score.comment == "This is a comment" + assert score.subject.kind == "observation" + assert score.subject.id == span_id + assert score.data_type == "NUMERIC" def test_create_trace_and_span(): @@ -963,16 +876,15 @@ def test_create_trace_and_span(): # Ensure data is sent langfuse.flush() - sleep(2) # Retrieve and verify - trace = get_api().trace.get(trace_id) + observations = wait_for_observations(trace_id, min_count=2) - assert trace.name == trace_name - assert len(trace.observations) == 2 # Parent span and child span + assert get_root_observation(observations).trace_name == trace_name + assert len(observations) == 2 # Parent span and child span # Find the child span - child_spans = [obs for obs in trace.observations if obs.name == "span"] + child_spans = [obs for obs in observations if obs.name == "span"] assert len(child_spans) == 1 span = child_spans[0] @@ -1004,35 +916,30 @@ def test_create_trace_and_generation(): # Ensure data is sent langfuse.flush() - sleep(2) - # Retrieve traces in two ways - dbTrace = get_api().trace.get(trace_id) - getTrace = get_api().trace.get( - trace_id - ) # Using API as direct getTrace not available + # Retrieve and verify + observations = wait_for_observations(trace_id, min_count=2) + root = get_root_observation(observations) # Verify trace details - assert dbTrace.name == trace_name - assert len(dbTrace.observations) == 2 # Parent span and generation - assert getTrace.name == trace_name - assert len(getTrace.observations) == 2 - assert getTrace.session_id == "test-session-id" + assert root.trace_name == trace_name + assert len(observations) == 2 # Parent span and generation + assert root.session_id == "test-session-id" # Find the generation - generations = [obs for obs in getTrace.observations if obs.name == "generation"] + generations = [obs for obs in observations if obs.name == "generation"] assert len(generations) == 1 generation = generations[0] assert generation.name == "generation" assert generation.trace_id == trace_id + assert generation.session_id == "test-session-id" assert generation.start_time is not None - assert getTrace.input == {"key": "value"} + assert root.input == {"key": "value"} def test_create_generation_and_trace(): langfuse = Langfuse() - api_wrapper = LangfuseAPI() trace_name = create_uuid() @@ -1063,23 +970,24 @@ def test_create_generation_and_trace(): # Ensure data is sent langfuse.flush() - sleep(2) # Retrieve and verify - trace = api_wrapper.get_trace(trace_id) - - assert trace["name"] == trace_name + observations = wait_for_observations(trace_id, min_count=2) # We should have 2 observations (the generation and the span for updating trace) - assert len(trace["observations"]) == 2 + assert len(observations) == 2 + + trace_update_spans = [obs for obs in observations if obs.name == "trace-update"] + assert len(trace_update_spans) == 1 + assert trace_update_spans[0].trace_name == trace_name # Find the generation - generations = [obs for obs in trace["observations"] if obs["name"] == "generation"] + generations = [obs for obs in observations if obs.name == "generation"] assert len(generations) == 1 generation_obs = generations[0] - assert generation_obs["name"] == "generation" - assert generation_obs["traceId"] == trace["id"] + assert generation_obs.name == "generation" + assert generation_obs.trace_id == trace_id def test_create_span_and_get_observation(): @@ -1096,12 +1004,12 @@ def test_create_span_and_get_observation(): # Flush and wait langfuse.flush() - sleep(2) - # Use API to fetch the observation by ID - observation = get_api().legacy.observations_v1.get(span_id) + observations = wait_for_observations(span.trace_id) # Verify observation properties + assert len(observations) == 1 + observation = observations[0] assert observation.name == "span" assert observation.id == span_id @@ -1123,20 +1031,19 @@ def test_update_generation(): # Ensure data is sent langfuse.flush() - sleep(2) # Retrieve and verify - trace = get_api().trace.get(trace_id) + observations = wait_for_observations(trace_id) # Verify trace properties - assert trace.name == "generation" - assert len(trace.observations) == 1 + assert len(observations) == 1 # Verify generation updates - retrieved_generation = trace.observations[0] + retrieved_generation = observations[0] + assert retrieved_generation.trace_name == "generation" assert retrieved_generation.name == "generation" assert retrieved_generation.trace_id == trace_id - assert retrieved_generation.metadata["dict"] == "value" + assert user_metadata(retrieved_generation) == {"dict": "value"} # Note: With OTEL, we can't verify exact start times from manually set timestamps, # as they are managed internally by the OTEL SDK @@ -1159,20 +1066,19 @@ def test_update_span(): # Ensure data is sent langfuse.flush() - sleep(2) # Retrieve and verify - trace = get_api().trace.get(trace_id) + observations = wait_for_observations(trace_id) # Verify trace properties - assert trace.name == "span" - assert len(trace.observations) == 1 + assert len(observations) == 1 # Verify span updates - retrieved_span = trace.observations[0] + retrieved_span = observations[0] + assert retrieved_span.trace_name == "span" assert retrieved_span.name == "span" assert retrieved_span.trace_id == trace_id - assert retrieved_span.metadata["dict"] == "value" + assert user_metadata(retrieved_span) == {"dict": "value"} def test_create_span_and_generation(): @@ -1197,17 +1103,16 @@ def test_create_span_and_generation(): # Ensure data is sent langfuse.flush() - sleep(2) # Retrieve and verify - trace = get_api().trace.get(trace_id) + observations = wait_for_observations(trace_id, min_count=2) # Verify trace details - assert len(trace.observations) == 2 + assert len(observations) == 2 # Find span and generation - spans = [obs for obs in trace.observations if obs.name == "span"] - generations = [obs for obs in trace.observations if obs.name == "generation"] + spans = [obs for obs in observations if obs.name == "span"] + generations = [obs for obs in observations if obs.name == "generation"] assert len(spans) == 1 assert len(generations) == 1 @@ -1222,7 +1127,6 @@ def test_create_span_and_generation(): def test_create_trace_with_id_and_generation(): langfuse = Langfuse() - api_wrapper = LangfuseAPI() trace_name = create_uuid() @@ -1244,28 +1148,27 @@ def test_create_trace_with_id_and_generation(): # Ensure data is sent langfuse.flush() - sleep(2) # Retrieve and verify - trace = api_wrapper.get_trace(trace_id) + observations = wait_for_observations(trace_id, min_count=2) # Verify trace properties - assert trace["name"] == trace_name - assert trace["id"] == trace_id - assert len(trace["observations"]) == 2 # Parent span and generation + root = get_root_observation(observations) + assert root.trace_name == trace_name + assert root.trace_id == trace_id + assert len(observations) == 2 # Parent span and generation # Find the generation - generations = [obs for obs in trace["observations"] if obs["name"] == "generation"] + generations = [obs for obs in observations if obs.name == "generation"] assert len(generations) == 1 gen = generations[0] - assert gen["name"] == "generation" - assert gen["traceId"] == trace["id"] + assert gen.name == "generation" + assert gen.trace_id == trace_id def test_end_generation(): langfuse = Langfuse() - api_wrapper = LangfuseAPI() # Create a generation generation = langfuse.start_observation( @@ -1294,21 +1197,14 @@ def test_end_generation(): langfuse.flush() # Retrieve and verify - trace = api_wrapper.get_trace( - trace_id, - is_result_ready=lambda trace: any( - obs["name"] == "query-generation" for obs in trace.get("observations", []) - ), - ) + observations = wait_for_observations(trace_id) # Find generation by name - generations = [ - obs for obs in trace["observations"] if obs["name"] == "query-generation" - ] + generations = [obs for obs in observations if obs.name == "query-generation"] assert len(generations) == 1 gen = generations[0] - assert gen["endTime"] is not None + assert gen.end_time is not None def test_end_generation_with_data(): @@ -1353,15 +1249,12 @@ def test_end_generation_with_data(): # Ensure data is sent langfuse.flush() - sleep(2) # Retrieve and verify - fetched_trace = get_api().trace.get(trace_id) + observations = wait_for_observations(trace_id, min_count=2) # Find generation by name - generations = [ - obs for obs in fetched_trace.observations if obs.name == "query-generation" - ] + generations = [obs for obs in observations if obs.name == "query-generation"] assert len(generations) == 1 generation = generations[0] @@ -1371,7 +1264,7 @@ def test_end_generation_with_data(): 2023, 1, 1, 12, 3, tzinfo=timezone.utc ) assert generation.name == "query-generation" - assert generation.metadata["dict"] == "value" + assert user_metadata(generation) == {"dict": "value"} assert generation.level == "ERROR" assert generation.status_message == "Generation ended" assert generation.version == "1.0" @@ -1379,12 +1272,9 @@ def test_end_generation_with_data(): assert generation.model_parameters == {"param1": "value1", "param2": "value2"} assert generation.input == [{"test_input_key": "test_input_value"}] assert generation.output == {"test_output_key": "test_output_value"} - assert generation.usage.input == 100 - assert generation.usage.output == 200 - assert generation.usage.total == 500 - assert generation.calculated_input_cost == 111 - assert generation.calculated_output_cost == 222 - assert generation.calculated_total_cost == 444 + assert generation.usage_details == {"input": 100, "output": 200, "total": 500} + assert generation.cost_details == {"input": 111, "output": 222, "total": 444} + assert generation.total_cost == 444 def test_end_generation_with_openai_token_format(): @@ -1418,31 +1308,26 @@ def test_end_generation_with_openai_token_format(): # Ensure data is sent langfuse.flush() - sleep(2) # Retrieve and verify - trace = get_api().trace.get(trace_id) + observations = wait_for_observations(trace_id) # Find generation - generations = [obs for obs in trace.observations if obs.name == "query-generation"] + generations = [obs for obs in observations if obs.name == "query-generation"] assert len(generations) == 1 generation_api = generations[0] # Verify properties were converted correctly assert generation_api.end_time is not None - assert generation_api.usage.input == 100 # prompt_tokens mapped to input - assert generation_api.usage.output == 200 # completion_tokens mapped to output - assert generation_api.usage.total == 500 - assert generation_api.usage.unit == "TOKENS" # Default unit for OpenAI format - assert generation_api.calculated_input_cost == 111 - assert generation_api.calculated_output_cost == 222 - assert generation_api.calculated_total_cost == 444 + # OpenAI-style keys are mapped to input/output/total + assert generation_api.usage_details == {"input": 100, "output": 200, "total": 500} + assert generation_api.cost_details == {"input": 111, "output": 222, "total": 444} + assert generation_api.total_cost == 444 def test_end_span(): langfuse = Langfuse() - api_wrapper = LangfuseAPI() # Create a span span = langfuse.start_observation( @@ -1460,19 +1345,18 @@ def test_end_span(): # Ensure data is sent langfuse.flush() - sleep(2) # Retrieve and verify - trace = api_wrapper.get_trace(trace_id) + observations = wait_for_observations(trace_id) # Find span - spans = [obs for obs in trace["observations"] if obs["name"] == "span"] + spans = [obs for obs in observations if obs.name == "span"] assert len(spans) == 1 span_api = spans[0] # Verify end time was set - assert span_api["endTime"] is not None + assert span_api.end_time is not None def test_end_span_with_data(): @@ -1495,21 +1379,19 @@ def test_end_span_with_data(): # Ensure data is sent langfuse.flush() - sleep(2) # Retrieve and verify - trace = get_api().trace.get(trace_id) + observations = wait_for_observations(trace_id) # Find span - spans = [obs for obs in trace.observations if obs.name == "span"] + spans = [obs for obs in observations if obs.name == "span"] assert len(spans) == 1 span_api = spans[0] # Verify end time and metadata were updated assert span_api.end_time is not None - assert span_api.metadata["dict"] == "value" - assert span_api.metadata["interface"] == "whatsapp" + assert user_metadata(span_api) == {"dict": "value", "interface": "whatsapp"} def test_get_generations(): @@ -1535,16 +1417,15 @@ def test_get_generations(): # Ensure data is sent langfuse.flush() - sleep(3) # Fetch generations using API - generations = get_api().legacy.observations_v1.get_many(name=generation_name) + generations = wait_for_observations(name=generation_name) # Verify fetched generation matches what we created - assert len(generations.data) == 1 - assert generations.data[0].name == generation_name - assert generations.data[0].input == "great-prompt" - assert generations.data[0].output == "great-completion" + assert len(generations) == 1 + assert generations[0].name == generation_name + assert generations[0].input == "great-prompt" + assert generations[0].output == "great-completion" def test_get_generations_by_user(): @@ -1574,18 +1455,15 @@ def test_get_generations_by_user(): # Ensure data is sent langfuse.flush() - sleep(3) # Fetch generations by user ID using the API - generations = get_api().legacy.observations_v1.get_many( - user_id=user_id, type="GENERATION" - ) + generations = wait_for_observations(user_id=user_id, type="GENERATION") # Verify fetched generation matches what we created - assert len(generations.data) == 1 - assert generations.data[0].name == generation_name - assert generations.data[0].input == "great-prompt" - assert generations.data[0].output == "great-completion" + assert len(generations) == 1 + assert generations[0].name == generation_name + assert generations[0].input == "great-prompt" + assert generations[0].output == "great-completion" def test_kwargs(): @@ -1614,16 +1492,17 @@ def test_kwargs(): # Ensure data is sent langfuse.flush() - sleep(2) # Retrieve and verify - observation = get_api().legacy.observations_v1.get(span_id) + observations = wait_for_observations(span.trace_id) + assert [observation.id for observation in observations] == [span_id] + observation = observations[0] # Verify kwargs were properly set as attributes assert observation.start_time is not None assert observation.input == {"key": "value"} assert observation.output == {"key": "value"} - assert observation.metadata["interface"] == "whatsapp" + assert user_metadata(observation) == {"interface": "whatsapp"} @pytest.mark.skip("Flaky") @@ -1660,16 +1539,13 @@ def test_timezone_awareness(): # Ensure data is sent langfuse.flush() - sleep(2) # Retrieve and verify - trace = get_api().trace.get(trace_id) + observations = wait_for_observations(trace_id, min_count=4) # Verify timestamps are in UTC regardless of local timezone - assert ( - len(trace.observations) == 4 - ) # Parent span, child span, generation, and event - for observation in trace.observations: + assert len(observations) == 4 # Parent span, child span, generation, and event + for observation in observations: # Check that start_time is within 5 seconds of current time delta = observation.start_time - utc_now assert delta.seconds < 5 @@ -1720,16 +1596,13 @@ def test_timezone_awareness_setting_timestamps(): # Ensure data is sent langfuse.flush() - sleep(2) # Retrieve and verify - trace = get_api().trace.get(trace_id) + observations = wait_for_observations(trace_id, min_count=4) # Verify timestamps are in UTC regardless of local timezone - assert ( - len(trace.observations) == 4 - ) # Parent span, child span, generation, and event - for observation in trace.observations: + assert len(observations) == 4 # Parent span, child span, generation, and event + for observation in observations: # Check that start_time is within 5 seconds of current time delta = abs((utc_now - observation.start_time).total_seconds()) assert delta < 5 @@ -1762,17 +1635,17 @@ def test_get_trace_by_session_id(): # Ensure data is sent langfuse.flush() - sleep(2) - # Retrieve the trace using the session_id - traces = get_api().trace.list(session_id=session_id) + # Retrieve the trace's observations using the session_id + observations = wait_for_observations(session_id=session_id) # Verify that the trace was retrieved correctly - assert len(traces.data) == 1 - retrieved_trace = traces.data[0] - assert retrieved_trace.name == trace_name - assert retrieved_trace.session_id == session_id - assert retrieved_trace.id == trace_id + assert len(observations) == 1 + retrieved_root = observations[0] + assert retrieved_root.is_root_observation is True + assert retrieved_root.trace_name == trace_name + assert retrieved_root.session_id == session_id + assert retrieved_root.trace_id == trace_id def test_fetch_trace(): @@ -1789,15 +1662,12 @@ def test_fetch_trace(): # Ensure data is sent langfuse.flush() - sleep(2) - # Fetch the trace using the get_api client - # Note: In the OTEL-based client, we use the API client directly - trace = get_api().trace.get(trace_id) + root = wait_for_root_observation(trace_id) # Verify trace properties - assert trace.id == trace_id - assert trace.name == name + assert root.trace_id == trace_id + assert root.trace_name == name def test_fetch_traces(): @@ -1836,38 +1706,50 @@ def test_fetch_traces(): expected_trace_ids = set(trace_ids) api = get_api(retry=False) + trace_name_filter = json.dumps( + [{"type": "string", "column": "traceName", "operator": "=", "value": name}] + ) - # Fetch all traces with the same name. - all_traces = wait_for_result( - lambda: api.trace.list(name=name, limit=10), - is_result_ready=lambda response: ( - {trace.id for trace in response.data} == expected_trace_ids + # Fetch the root observations of all traces with the same name. + roots = wait_for_observations( + filter=trace_name_filter, + is_result_ready=lambda observations: ( + {o.trace_id for o in observations} == expected_trace_ids ), ) # Verify we got all traces - assert len(all_traces.data) == 3 + assert len(roots) == 3 # Verify trace properties - for trace in all_traces.data: - assert trace.name == name - assert trace.session_id == "session-1" - assert trace.input == {"key": "value"} - assert trace.output == "output-value" - - # Test pagination by fetching the first three pages one at a time and - # confirming they collectively cover the created traces. - paginated_ids = set() - for page in range(1, 4): - paginated_response = wait_for_result( - lambda page=page: api.trace.list(name=name, limit=1, page=page), - is_result_ready=lambda response: ( - len(response.data) == 1 and response.data[0].id in expected_trace_ids - ), + for root in roots: + assert root.is_root_observation is True + assert root.trace_name == name + assert root.session_id == "session-1" + assert root.input == {"key": "value"} + assert root.output == "output-value" + + # Test cursor pagination by walking pages of one item and confirming they + # collectively cover the created traces. + paginated_ids = [] + cursor = None + for _ in range(3): + page = api.observations.get_many( + filter=trace_name_filter, limit=1, cursor=cursor + ) + assert len(page.data) == 1 + paginated_ids.append(page.data[0].trace_id) + cursor = page.meta.cursor + + assert set(paginated_ids) == expected_trace_ids + assert len(paginated_ids) == 3 + if cursor is not None: + assert ( + api.observations.get_many( + filter=trace_name_filter, limit=1, cursor=cursor + ).data + == [] ) - paginated_ids.add(paginated_response.data[0].id) - - assert paginated_ids == expected_trace_ids def test_get_observation(): @@ -1890,10 +1772,12 @@ def test_get_observation(): # Ensure data is sent langfuse.flush() - sleep(2) # Fetch the observation using the API - observation = get_api().legacy.observations_v1.get(generation_id) + observations = wait_for_observations(parent_span.trace_id, min_count=2) + matching = [o for o in observations if o.id == generation_id] + assert len(matching) == 1 + observation = matching[0] # Verify observation properties assert observation.id == generation_id @@ -1926,18 +1810,18 @@ def test_get_observations(): # Fetch observations using the API expected_generation_ids = {gen1_id, gen2_id} - observations = wait_for_result( - lambda: api.legacy.observations_v1.get_many(name=name, limit=10), - is_result_ready=lambda response: expected_generation_ids.issubset( - {obs.id for obs in response.data} + observations = wait_for_observations( + name=name, + is_result_ready=lambda observations: expected_generation_ids.issubset( + {obs.id for obs in observations} ), ) # Verify fetched observations - assert len(observations.data) == 2 + assert len(observations) == 2 # Filter for just the generations - generations = [obs for obs in observations.data if obs.type == "GENERATION"] + generations = [obs for obs in observations if obs.type == "GENERATION"] assert len(generations) == 2 # Verify the generation IDs match what we created @@ -1945,91 +1829,32 @@ def test_get_observations(): assert gen1_id in gen_ids assert gen2_id in gen_ids - # Test pagination by confirming both created generations can be reached - # across separate pages. - paginated_ids = set() - for page in range(1, 3): - paginated_response = wait_for_result( - lambda page=page: api.legacy.observations_v1.get_many( - name=name, limit=1, page=page - ), - is_result_ready=lambda response: ( - len(response.data) == 1 - and response.data[0].id in expected_generation_ids - ), - ) - paginated_ids.add(paginated_response.data[0].id) - - assert paginated_ids == expected_generation_ids - - -def test_get_trace_not_found(): - # Attempt to fetch a non-existent trace using the API - with pytest.raises(Exception): - get_api(retry=False).trace.get(create_uuid()) - + # Test cursor pagination by confirming both created generations can be + # reached across separate pages. + first_page = api.observations.get_many(name=name, limit=1) + assert len(first_page.data) == 1 + assert first_page.meta.cursor is not None + second_page = api.observations.get_many( + name=name, limit=1, cursor=first_page.meta.cursor + ) + assert len(second_page.data) == 1 -def test_get_observation_not_found(): - # Attempt to fetch a non-existent observation using the API - with pytest.raises(Exception): - get_api(retry=False).legacy.observations_v1.get(create_uuid()) + assert {first_page.data[0].id, second_page.data[0].id} == expected_generation_ids -def test_get_traces_empty(): - # Fetch traces with a filter that should return no results - response = get_api(retry=False).trace.list(name=create_uuid()) +def test_get_observations_for_unknown_trace_is_empty(): + response = get_api(retry=False).observations.get_many(trace_id=create_uuid()) - assert len(response.data) == 0 - assert response.meta.total_items == 0 + assert response.data == [] + assert response.meta.cursor is None def test_get_observations_empty(): # Fetch observations with a filter that should return no results - response = get_api(retry=False).legacy.observations_v1.get_many(name=create_uuid()) + response = get_api(retry=False).observations.get_many(name=create_uuid()) - assert len(response.data) == 0 - assert response.meta.total_items == 0 - - -def test_get_sessions(): - langfuse = Langfuse() - - # unique name - name = create_uuid() - session1 = create_uuid() - session2 = create_uuid() - session3 = create_uuid() - - # Create multiple traces with different session IDs - # Create first trace - with langfuse.start_as_current_observation(name=name): - with propagate_attributes(trace_name=name, session_id=session1): - pass - - # Create second trace - with langfuse.start_as_current_observation(name=name): - with propagate_attributes(trace_name=name, session_id=session2): - pass - - # Create third trace - with langfuse.start_as_current_observation(name=name): - with propagate_attributes(trace_name=name, session_id=session3): - pass - - langfuse.flush() - - # Fetch sessions - sleep(3) - response = get_api().sessions.list() - - # Assert the structure of the response, cannot check for the exact number of sessions as the table is not cleared between tests - assert hasattr(response, "data") - assert hasattr(response, "meta") - assert isinstance(response.data, list) - - # fetch only one, cannot check for the exact number of sessions as the table is not cleared between tests - response = get_api().sessions.list(limit=1, page=2) - assert len(response.data) == 1 + assert response.data == [] + assert response.meta.cursor is None @pytest.mark.skip( @@ -2037,7 +1862,6 @@ def test_get_sessions(): ) def test_create_trace_sampling_zero(): langfuse = Langfuse(sample_rate=0) - api_wrapper = LangfuseAPI() trace_name = create_uuid() # Create a span with trace properties - with sample_rate=0, this will not be sent to the API @@ -2061,12 +1885,9 @@ def test_create_trace_sampling_zero(): langfuse.flush() sleep(2) - # Try to fetch the trace - should fail as it wasn't sent to the API - fetched_trace = api_wrapper.get_trace(trace_id) - assert fetched_trace == { - "error": "LangfuseNotFoundError", - "message": f"Trace {trace_id} not found within authorized project", - } + # The trace's observations must not exist as they were never sent to the API + assert get_observations(trace_id=trace_id) == [] + assert get_scores(trace_id=trace_id) == [] def test_mask_function(request): @@ -2083,7 +1904,6 @@ def mask_func(data): return data langfuse = Langfuse(mask=mask_func) - api_wrapper = LangfuseAPI() # Create a root span with trace properties with langfuse.start_as_current_observation(name="test-span") as root_span: @@ -2112,26 +1932,22 @@ def mask_func(data): # Ensure data is sent langfuse.flush() - sleep(2) # Retrieve and verify - fetched_trace = api_wrapper.get_trace(trace_id) - assert fetched_trace["input"] == {"sensitive": "MASKED"} - assert fetched_trace["output"] == {"more": "MASKED"} + observations = wait_for_observations(trace_id, min_count=3) + fetched_root = get_root_observation(observations) + assert fetched_root.input == {"sensitive": "MASKED"} + assert fetched_root.output == {"more": "MASKED"} - fetched_gen = [ - o for o in fetched_trace["observations"] if o["type"] == "GENERATION" - ][0] - assert fetched_gen["input"] == {"prompt": "MASKED"} - assert fetched_gen["output"] == "MASKED" + fetched_gen = [o for o in observations if o.type == "GENERATION"][0] + assert fetched_gen.input == {"prompt": "MASKED"} + assert fetched_gen.output == "MASKED" fetched_span = [ - o - for o in fetched_trace["observations"] - if o["type"] == "SPAN" and o["name"] == "test_span" + o for o in observations if o.type == "SPAN" and o.name == "test_span" ][0] - assert fetched_span["input"] == {"data": "MASKED"} - assert fetched_span["output"] == "MASKED" + assert fetched_span.input == {"data": "MASKED"} + assert fetched_span.output == "MASKED" # Create a root span with trace properties with langfuse.start_as_current_observation(name="test-span") as root_span: @@ -2144,12 +1960,11 @@ def mask_func(data): # Ensure data is sent langfuse.flush() - sleep(2) # Retrieve and verify - fetched_trace = api_wrapper.get_trace(trace_id) - assert fetched_trace["input"] == "" - assert fetched_trace["output"] == "" + fetched_root = wait_for_root_observation(trace_id) + assert fetched_root.input == "" + assert fetched_root.output == "" def test_get_project_id(): @@ -2218,13 +2033,11 @@ def test_start_as_current_observation_types(): pass langfuse.flush() - sleep(2) - api = get_api() - trace = api.trace.get(trace_id) + observations = wait_for_observations(trace_id, min_count=len(observation_types) + 1) # Check we have all expected observation types - found_types = {obs.type for obs in trace.observations} + found_types = {obs.type for obs in observations} expected_types = {obs_type.upper() for obs_type in observation_types} | { "SPAN" } # includes parent span @@ -2234,12 +2047,12 @@ def test_start_as_current_observation_types(): # Verify each specific observation exists for obs_type in observation_types: - observations = [ + matching = [ obs - for obs in trace.observations + for obs in observations if obs.name == f"test-{obs_type}" and obs.type == obs_type.upper() ] - assert len(observations) == 1, f"Expected one {obs_type.upper()} observation" + assert len(matching) == 1, f"Expected one {obs_type.upper()} observation" def test_that_generation_like_properties_are_actually_created(): @@ -2296,21 +2109,22 @@ def test_that_generation_like_properties_are_actually_created(): langfuse.flush() - api = get_api() - trace = api.trace.get(trace_id) + observations = wait_for_observations( + trace_id, min_count=len(generation_like_types) + 1 + ) # Verify that the properties are persisted in the API for generation-like types for obs_type in generation_like_types: - observations = [ + matching = [ obs - for obs in trace.observations + for obs in observations if obs.name == f"test-{obs_type}" and obs.type == obs_type.upper() ] - assert len(observations) == 1, ( - f"Expected one {obs_type.upper()} observation, but found {len(observations)}" + assert len(matching) == 1, ( + f"Expected one {obs_type.upper()} observation, but found {len(matching)}" ) - obs = observations[0] + obs = matching[0] assert obs.model == test_model, f"{obs_type} should have model property" assert obs.model_parameters == test_model_parameters, ( diff --git a/tests/e2e/test_decorators.py b/tests/e2e/test_decorators.py index acd495f57..a33a4f3d5 100644 --- a/tests/e2e/test_decorators.py +++ b/tests/e2e/test_decorators.py @@ -15,7 +15,11 @@ from langfuse._client.resource_manager import LangfuseResourceManager from langfuse.langchain import CallbackHandler from langfuse.media import LangfuseMedia -from tests.support.utils import get_api, wait_for_trace +from tests.support.utils import ( + get_observations, + user_metadata, + wait_for_trace_snapshot, +) mock_metadata = {"key": "metadata"} mock_deep_metadata = {"key": "mock_deep_metadata"} @@ -100,7 +104,7 @@ def level_1_function(*args, **kwargs): assert result == "level_1" # Wrapped function returns correctly # ID setting for span or trace - trace_data = get_api().trace.get(mock_trace_id) + trace_data = wait_for_trace_snapshot(mock_trace_id, min_observations=3) assert len(trace_data.observations) == 3 # trace parameters if set anywhere in the call stack @@ -133,7 +137,7 @@ def level_1_function(*args, **kwargs): assert level_3_observation.name == "level_3" assert level_3_observation.metadata["key"] == mock_deep_metadata["key"] assert level_3_observation.type == "GENERATION" - assert level_3_observation.calculated_total_cost > 0 + assert level_3_observation.total_cost > 0 assert level_3_observation.output == "mock_output" assert level_3_observation.version == "version-1" @@ -182,7 +186,7 @@ def level_1_function(*args, **kwargs): assert result == "level_1" # Wrapped function returns correctly # ID setting for span or trace - trace_data = get_api().trace.get(mock_trace_id) + trace_data = wait_for_trace_snapshot(mock_trace_id, min_observations=3) assert len(trace_data.observations) == 3 # trace parameters if set anywhere in the call stack @@ -215,7 +219,7 @@ def level_1_function(*args, **kwargs): assert level_3_observation.name == "level_3" assert level_3_observation.metadata["key"] == mock_deep_metadata["key"] assert level_3_observation.type == "GENERATION" - assert level_3_observation.calculated_total_cost > 0 + assert level_3_observation.total_cost > 0 assert level_3_observation.output == "mock_output" assert level_3_observation.version == "version-1" @@ -260,7 +264,7 @@ def level_1_function(*args, **kwargs): langfuse.flush() - trace_data = get_api().trace.get(mock_trace_id) + trace_data = wait_for_trace_snapshot(mock_trace_id, min_observations=3) # trace parameters if set anywhere in the call stack assert trace_data.session_id == mock_session_id @@ -349,7 +353,7 @@ def level_1_function(*args, **kwargs): langfuse.flush() for mock_id in [mock_trace_id_1, mock_trace_id_2]: - trace_data = get_api().trace.get(mock_id) + trace_data = wait_for_trace_snapshot(mock_id, min_observations=3) assert len(trace_data.observations) == 3 # ID setting for span or trace @@ -382,7 +386,7 @@ def level_1_function(*args, **kwargs): assert level_3_observation.metadata["key"] == mock_deep_metadata["key"] assert level_3_observation.type == "GENERATION" - assert level_3_observation.calculated_total_cost > 0 + assert level_3_observation.total_cost > 0 def test_decorators_langchain(): @@ -427,7 +431,7 @@ def level_1_function(*args, **kwargs): langfuse.flush() - trace_data = wait_for_trace( + trace_data = wait_for_trace_snapshot( mock_trace_id, is_result_ready=lambda trace: ( trace.session_id == mock_session_id @@ -522,8 +526,10 @@ def level_1_function(*args, **kwargs): assert result == "level_3" # Wrapped function returns correctly # ID setting for span or trace - trace_data = wait_for_trace( + trace_data = wait_for_trace_snapshot( mock_trace_id, + min_observations=3, + min_scores=3, is_result_ready=lambda trace: { "test-observation-score", "test-trace-score", @@ -553,7 +559,7 @@ def level_1_function(*args, **kwargs): assert any( [ score.name == "another-test-trace-score" - and score.string_value == "my_value" + and score.value == "my_value" and score.data_type == "CATEGORICAL" for score in trace_scores ] @@ -599,7 +605,7 @@ def function_with_circular_arg(circular_obj, *args, **kwargs): # Validate that the function executed as expected assert result == "function response" - trace_data = get_api().trace.get(mock_trace_id) + trace_data = wait_for_trace_snapshot(mock_trace_id, min_observations=1) assert ( trace_data.observations[0].input["args"][0]["reference"] == "CircularRefObject" @@ -632,7 +638,7 @@ def main(*args, **kwargs): assert result == "function response" - trace_data = get_api().trace.get(mock_trace_id) + trace_data = wait_for_trace_snapshot(mock_trace_id, min_observations=2) # Check that disabled capture_io doesn't capture manually set input/output assert len(trace_data.observations) == 2 @@ -702,7 +708,7 @@ def level_1_function(self, *args, **kwargs): assert result == "level_1" # Wrapped function returns correctly # ID setting for span or trace - trace_data = get_api().trace.get(mock_trace_id) + trace_data = wait_for_trace_snapshot(mock_trace_id, min_observations=4) assert len(trace_data.observations) == 4 # trace parameters if set anywhere in the call stack @@ -743,7 +749,7 @@ def level_1_function(self, *args, **kwargs): assert level_3_observation.name == "level_3_function" assert level_3_observation.metadata["key"] == mock_deep_metadata["key"] assert level_3_observation.type == "GENERATION" - assert level_3_observation.calculated_total_cost > 0 + assert level_3_observation.total_cost > 0 assert level_3_observation.output == "mock_output" @@ -779,7 +785,7 @@ def main(**kwargs): assert result == mock_output - trace_data = get_api().trace.get(mock_trace_id) + trace_data = wait_for_trace_snapshot(mock_trace_id, min_observations=2) # Find the main and nested observations adjacencies = defaultdict(list) @@ -833,7 +839,7 @@ async def main_async(**kwargs): assert result == mock_output - trace_data = get_api().trace.get(mock_trace_id) + trace_data = wait_for_trace_snapshot(mock_trace_id, min_observations=2) # Check correct nesting adjacencies = defaultdict(list) @@ -902,7 +908,7 @@ async def level_1_function(*args, **kwargs): assert result == "level_1" # Wrapped function returns correctly # ID setting for span or trace - trace_data = wait_for_trace( + trace_data = wait_for_trace_snapshot( mock_trace_id, is_result_ready=lambda trace: ( trace.session_id == mock_session_id @@ -946,11 +952,11 @@ async def level_1_function(*args, **kwargs): "max_tokens": "Infinity", "presence_penalty": 0, } - assert generation.usage.input is not None - assert generation.usage.output is not None - assert generation.usage.total is not None + assert generation.usage_details["input"] is not None + assert generation.usage_details["output"] is not None + assert generation.usage_details["total"] is not None print(generation) - assert generation.output == 2 + assert generation.output == "2" def test_generator_as_function_input(): @@ -982,7 +988,7 @@ def main(**kwargs): assert result == mock_output - trace_data = get_api().trace.get(mock_trace_id) + trace_data = wait_for_trace_snapshot(mock_trace_id, min_observations=2) nested_obs = next(o for o in trace_data.observations if o.name == "nested") @@ -1019,7 +1025,7 @@ def main(**kwargs): main(langfuse_trace_id=mock_trace_id) langfuse.flush() - trace_data = get_api().trace.get(mock_trace_id) + trace_data = wait_for_trace_snapshot(mock_trace_id, min_observations=2) # Find the observation with name 'nested' nested_observation = next(o for o in trace_data.observations if o.name == "nested") @@ -1052,7 +1058,7 @@ def function(): assert result == mock_output - trace_data = wait_for_trace( + trace_data = wait_for_trace_snapshot( mock_trace_id, is_result_ready=lambda trace: any( observation.name == "function" and observation.output == mock_output @@ -1099,7 +1105,7 @@ def main(): langfuse.flush() - trace_data = wait_for_trace( + trace_data = wait_for_trace_snapshot( mock_trace_id, is_result_ready=lambda trace: ( "@@@langfuseMedia:type=application/pdf|id=" @@ -1155,19 +1161,20 @@ def main(): langfuse.flush() - trace_data = wait_for_trace( + trace_data = wait_for_trace_snapshot( mock_trace_id, - is_result_ready=lambda trace: ( - trace.metadata is not None - and trace.metadata.get("key1") == "value1" - and trace.metadata.get("key2") == "value2" - and trace.tags == ["tag1", "tag2"] - ), + min_observations=2, + is_result_ready=lambda trace: trace.tags == ["tag1", "tag2"], ) - assert trace_data.metadata["key1"] == "value1" - assert trace_data.metadata["key2"] == "value2" + # Trace metadata comes from the root; the nested propagation only reaches + # the nested observation. + assert trace_data.metadata == {"key1": "value1"} + nested_observation = _get_observation_by_name(trace_data, "nested") + assert user_metadata(nested_observation) == {"key1": "value1", "key2": "value2"} + assert nested_observation.tags == ["tag1", "tag2"] + assert trace_data.root.tags == ["tag1"] assert trace_data.tags == ["tag1", "tag2"] @@ -1227,7 +1234,7 @@ def level_1_function(*args, **kwargs): assert result == "level_1" # Verify trace was created properly - trace_data = wait_for_trace( + trace_data = wait_for_trace_snapshot( mock_trace_id, is_result_ready=lambda trace: ( trace.name == mock_name and len(trace.observations) == 3 @@ -1287,7 +1294,7 @@ def level_1_function(*args, **kwargs): assert result == "level_4" - trace_data = wait_for_trace( + trace_data = wait_for_trace_snapshot( mock_trace_id, is_result_ready=lambda trace: ( trace.name == mock_name @@ -1362,7 +1369,7 @@ def level_1_function(*args, **kwargs): assert result == "level_1" - trace_data = wait_for_trace( + trace_data = wait_for_trace_snapshot( mock_trace_id, is_result_ready=lambda trace: ( trace.name == mock_name and len(trace.observations) == 2 @@ -1417,13 +1424,7 @@ def level_1_function(*args, **kwargs): # Should skip tracing entirely in multi-project setup without public key # This is expected behavior to prevent cross-project data leakage - try: - trace_data = get_api().trace.get(mock_trace_id) - # If trace is found, it should have no observations (tracing was skipped) - assert len(trace_data.observations) == 0 - except Exception: - # Trace not found is also expected - tracing was completely disabled - pass + assert get_observations(trace_id=mock_trace_id) == [] # Reset instances to not leak to other test suites removeMockResourceManagerInstances() @@ -1486,7 +1487,7 @@ async def async_level_1_function(*args, **kwargs): assert result == "async_level_3" # Verify trace was created properly - trace_data = wait_for_trace( + trace_data = wait_for_trace_snapshot( mock_trace_id, is_result_ready=lambda trace: ( trace.name == mock_name @@ -1568,7 +1569,7 @@ async def async_level_1_function(*args, **kwargs): assert result == "sync_level_4" - trace_data = get_api().trace.get(mock_trace_id) + trace_data = wait_for_trace_snapshot(mock_trace_id, min_observations=4) assert len(trace_data.observations) == 4 assert trace_data.name == mock_name @@ -1647,8 +1648,8 @@ async def async_level_1_function(task_id, *args, **kwargs): assert result2 == "async_level_3_task_2" # Verify both traces were created correctly and didn't interfere - trace_data_1 = get_api().trace.get(trace_id_1) - trace_data_2 = get_api().trace.get(trace_id_2) + trace_data_1 = wait_for_trace_snapshot(trace_id_1, min_observations=3) + trace_data_2 = wait_for_trace_snapshot(trace_id_2, min_observations=3) assert trace_data_1.name == f"{mock_name}_task_1" assert trace_data_2.name == f"{mock_name}_task_2" @@ -1721,7 +1722,7 @@ async def async_consumer_function(): assert result == "Hello, Async World!" - trace_data = get_api().trace.get(mock_trace_id) + trace_data = wait_for_trace_snapshot(mock_trace_id, min_observations=2) assert len(trace_data.observations) == 2 assert trace_data.name == mock_name @@ -1786,7 +1787,7 @@ async def async_root_function(*args, **kwargs): assert result == "exception_handled" - trace_data = get_api().trace.get(mock_trace_id) + trace_data = wait_for_trace_snapshot(mock_trace_id, min_observations=3) assert len(trace_data.observations) == 3 assert trace_data.name == mock_name @@ -1851,11 +1852,11 @@ def root_function(): ) # Verify trace structure - trace_data = wait_for_trace( + trace_data = wait_for_trace_snapshot( mock_trace_id, is_result_ready=lambda trace: ( len(trace.observations) >= 2 - and {"parent_root", "child_stream"}.issubset( + and {"root", "sync_generator"}.issubset( { observation.name for observation in trace.observations @@ -1928,7 +1929,7 @@ async def root_function(): ) # Verify trace structure - trace_data = get_api().trace.get(mock_trace_id) + trace_data = wait_for_trace_snapshot(mock_trace_id, min_observations=2) assert len(trace_data.observations) == 2 # Verify both observations are present @@ -1996,7 +1997,7 @@ async def parent_function(): ) # Verify trace structure - trace_data = get_api().trace.get(mock_trace_id) + trace_data = wait_for_trace_snapshot(mock_trace_id, min_observations=2) assert len(trace_data.observations) == 2 # Check both observations exist @@ -2043,7 +2044,7 @@ async def root_function(): assert items == ["first_item"] # Verify trace structure - should have both observations despite exception - trace_data = get_api().trace.get(mock_trace_id) + trace_data = wait_for_trace_snapshot(mock_trace_id, min_observations=2) assert len(trace_data.observations) == 2 # Check that the failing generator observation has ERROR level @@ -2083,7 +2084,7 @@ def root_function(): assert items == [] # Verify trace structure - trace_data = get_api().trace.get(mock_trace_id) + trace_data = wait_for_trace_snapshot(mock_trace_id, min_observations=2) assert len(trace_data.observations) == 2 # Verify empty generator observation diff --git a/tests/e2e/test_media.py b/tests/e2e/test_media.py index d322e1788..ad8a8cb37 100644 --- a/tests/e2e/test_media.py +++ b/tests/e2e/test_media.py @@ -4,7 +4,7 @@ from langfuse._client.client import Langfuse from langfuse.media import LangfuseMedia -from tests.support.utils import wait_for_trace +from tests.support.utils import wait_for_observations def test_replace_media_reference_string_in_object(): @@ -30,25 +30,26 @@ def test_replace_media_reference_string_in_object(): langfuse.flush() - fetched_trace = wait_for_trace( + fetched_observations = wait_for_observations( span.trace_id, - is_result_ready=lambda trace: ( - bool(trace.observations) - and re.match( + is_result_ready=lambda observations: ( + re.match( r"^@@@langfuseMedia:type=audio/wav\|id=.+\|source=base64_data_uri@@@$", - trace.observations[0].metadata.get("context", {}).get("nested", ""), + observations[0].metadata.get("context", {}).get("nested", ""), ) is not None ), ) - media_ref = fetched_trace.observations[0].metadata["context"]["nested"] + assert len(fetched_observations) == 1 + fetched_observation = fetched_observations[0] + media_ref = fetched_observation.metadata["context"]["nested"] assert re.match( r"^@@@langfuseMedia:type=audio/wav\|id=.+\|source=base64_data_uri@@@$", media_ref, ) resolved_obs = langfuse.resolve_media_references( - obj=fetched_trace.observations[0], resolve_with="base64_data_uri" + obj=fetched_observation, resolve_with="base64_data_uri" ) expected_base64 = f"data:audio/wav;base64,{base64_audio}" @@ -61,15 +62,11 @@ def test_replace_media_reference_string_in_object(): langfuse.flush() - fetched_trace2 = wait_for_trace( + fetched_observations2 = wait_for_observations( span2.trace_id, - is_result_ready=lambda trace: ( - bool(trace.observations) - and trace.observations[0].metadata.get("context", {}).get("nested") - == fetched_trace.observations[0].metadata["context"]["nested"] + is_result_ready=lambda observations: ( + observations[0].metadata.get("context", {}).get("nested") == media_ref ), ) - assert ( - fetched_trace2.observations[0].metadata["context"]["nested"] - == fetched_trace.observations[0].metadata["context"]["nested"] - ) + assert len(fetched_observations2) == 1 + assert fetched_observations2[0].metadata["context"]["nested"] == media_ref diff --git a/tests/e2e/test_prompt.py b/tests/e2e/test_prompt.py index 6e113cb41..5b611d1fb 100644 --- a/tests/e2e/test_prompt.py +++ b/tests/e2e/test_prompt.py @@ -1,7 +1,7 @@ import pytest from langfuse._client.client import Langfuse -from tests.support.utils import create_uuid, get_api +from tests.support.utils import create_uuid, wait_for_observations def test_create_prompt(): @@ -429,18 +429,17 @@ def test_prompt_end_to_end(): langfuse.flush() - api = get_api() - - trace = api.trace.get(generation.trace_id) - - assert len(trace.observations) == 1 - - generation = trace.observations[0] - assert generation.prompt_id is not None + observations = wait_for_observations( + generation.trace_id, + is_result_ready=lambda observations: observations[0].prompt_id is not None, + ) - observation = api.legacy.observations_v1.get(generation.id) + assert len(observations) == 1 + observation = observations[0] assert observation.prompt_id is not None + assert observation.prompt_name == "test" + assert observation.prompt_version == prompt.version def test_do_not_return_fallback_if_fetch_success(): @@ -525,11 +524,10 @@ def test_do_not_link_observation_if_fallback(): ).end() langfuse.flush() - api = get_api() - trace = api.trace.get(generation.trace_id) + observations = wait_for_observations(generation.trace_id) - assert len(trace.observations) == 1 - assert trace.observations[0].prompt_id is None + assert len(observations) == 1 + assert observations[0].prompt_id is None def test_variable_names_on_content_with_variable_names(): diff --git a/tests/live_provider/test_langchain.py b/tests/live_provider/test_langchain.py index d04ddb81c..c1ebe8566 100644 --- a/tests/live_provider/test_langchain.py +++ b/tests/live_provider/test_langchain.py @@ -1,3 +1,4 @@ +import json import random import string import time @@ -18,7 +19,12 @@ from langfuse._client.client import Langfuse from langfuse.langchain import CallbackHandler -from tests.support.utils import create_uuid, encode_file_to_base64, get_api +from tests.support.utils import ( + create_uuid, + encode_file_to_base64, + wait_for_observations, + wait_for_trace_snapshot, +) def test_callback_generated_from_trace_chat(): @@ -44,7 +50,7 @@ def test_callback_generated_from_trace_chat(): langfuse.flush() - trace = get_api().trace.get(trace_id) + trace = wait_for_trace_snapshot(trace_id, min_observations=2) assert trace.input is None assert trace.output is None @@ -92,7 +98,7 @@ def test_callback_generated_from_lcel_chain(): langfuse.flush() - trace = get_api().trace.get(trace_id) + trace = wait_for_trace_snapshot(trace_id, min_observations=2) assert trace.input is None assert trace.output is None @@ -151,10 +157,10 @@ def test_basic_chat_openai(): # Ensure data is flushed to API sleep(2) - # Retrieve trace by name - traces = get_api().trace.list(name=test_name) - assert len(traces.data) > 0 - trace = get_api().trace.get(traces.data[0].id) + # Retrieve trace by its root observation, which carries the run name + roots = wait_for_observations(name=test_name) + assert len(roots) > 0 + trace = wait_for_trace_snapshot(roots[0].trace_id, min_observations=2) # Assertions assert trace.name == test_name @@ -193,7 +199,7 @@ def test_callback_simple_openai(): sleep(2) # Retrieve trace - trace = get_api().trace.get(trace_id) + trace = wait_for_trace_snapshot(trace_id, min_observations=2) # Assertions - add 1 for the wrapping span assert len(trace.observations) > 1 @@ -241,7 +247,7 @@ def test_callback_multiple_invocations_on_different_traces(): sleep(2) # Retrieve trace - trace = get_api().trace.get(trace_id) + trace = wait_for_trace_snapshot(trace_id, min_observations=3) # Add 1 to account for the wrapping span assert len(trace.observations) > 2 @@ -300,7 +306,7 @@ def test_openai_instruct_usage(): lf_handler._langfuse_client.flush() - observations = get_api().trace.get(trace_id).observations + observations = wait_for_observations(trace_id, min_count=3) assert len(observations) >= 3 assert any( @@ -318,7 +324,7 @@ def test_openai_instruct_usage(): assert observation.output != "" assert observation.input is not None assert observation.input != "" - assert observation.usage is not None + assert observation.usage_details is not None assert observation.usage_details["input"] is not None assert observation.usage_details["output"] is not None assert observation.usage_details["total"] is not None @@ -479,7 +485,7 @@ def test_link_langfuse_prompts_invoke(): langfuse_handler._langfuse_client.flush() sleep(2) - trace = get_api().trace.get(trace_id=trace_id) + trace = wait_for_trace_snapshot(trace_id, min_observations=4) observations = trace.observations @@ -567,7 +573,12 @@ def test_link_langfuse_prompts_stream(): langfuse_handler._langfuse_client.flush() sleep(2) - trace = get_api().trace.get(trace_id=trace_id) + trace = wait_for_trace_snapshot( + trace_id, + is_result_ready=lambda trace: ( + len([o for o in trace.observations if o.type == "GENERATION"]) >= 4 + ), + ) observations = trace.observations @@ -653,11 +664,26 @@ def test_link_langfuse_prompts_batch(): langfuse_handler._langfuse_client.flush() - traces = get_api().trace.list(name=trace_name).data + trace_name_filter = json.dumps( + [ + { + "type": "string", + "column": "traceName", + "operator": "=", + "value": trace_name, + } + ] + ) + traced_observations = wait_for_observations(filter=trace_name_filter) - assert len(traces) == 1 + assert {o.trace_id for o in traced_observations} == {trace_id} - trace = get_api().trace.get(trace_id=trace_id) + trace = wait_for_trace_snapshot( + trace_id, + is_result_ready=lambda trace: ( + len([o for o in trace.observations if o.type == "GENERATION"]) >= 10 + ), + ) observations = trace.observations @@ -783,7 +809,7 @@ class GetWeather(BaseModel): handler._langfuse_client.flush() - trace = get_api().trace.get(trace_id=trace_id) + trace = wait_for_trace_snapshot(trace_id, min_observations=2) generations = list(filter(lambda x: x.type == "GENERATION", trace.observations)) assert len(generations) > 0 @@ -882,7 +908,7 @@ def test_multimodal(): handler._langfuse_client.flush() - trace = get_api().trace.get(trace_id=trace_id) + trace = wait_for_trace_snapshot(trace_id, min_observations=2) assert len(trace.observations) >= 2 assert any( @@ -979,7 +1005,7 @@ def call_model(state: MessagesState): print(final_state["messages"][-1].content) handler._langfuse_client.flush() - trace = get_api().trace.get(trace_id=trace_id) + trace = wait_for_trace_snapshot(trace_id, min_observations=2) assert len(trace.observations) > 0 @@ -1012,7 +1038,7 @@ def test_cached_token_usage(): handler._langfuse_client.flush() - trace = get_api().trace.get(handler.get_trace_id()) + trace = wait_for_trace_snapshot(handler.get_trace_id()) generation = next((o for o in trace.observations if o.type == "GENERATION")) diff --git a/tests/live_provider/test_langchain_integration.py b/tests/live_provider/test_langchain_integration.py index edb5455c4..dd6b166b2 100644 --- a/tests/live_provider/test_langchain_integration.py +++ b/tests/live_provider/test_langchain_integration.py @@ -7,7 +7,7 @@ from langfuse import Langfuse from langfuse.langchain import CallbackHandler -from tests.support.utils import create_uuid, get_api +from tests.support.utils import create_uuid, wait_for_trace_snapshot def _is_streaming_response(response): @@ -45,8 +45,7 @@ def test_stream_chat_models(model_name): langfuse_client.flush() assert handler.runs == {} - api = get_api() - trace = api.trace.get(trace_id) + trace = wait_for_trace_snapshot(trace_id, min_observations=2) generationList = list(filter(lambda o: o.type == "GENERATION", trace.observations)) assert len(generationList) != 0 @@ -61,15 +60,15 @@ def test_stream_chat_models(model_name): assert generation.model_parameters.get("max_completion_tokens") is not None assert generation.model_parameters.get("temperature") is not None assert generation.metadata["tags"] == tags - assert generation.usage.output is not None - assert generation.usage.total is not None + assert generation.usage_details["output"] is not None + assert generation.usage_details["total"] is not None assert generation.output["content"] is not None assert generation.output["role"] is not None assert generation.input_price is not None assert generation.output_price is not None - assert generation.calculated_input_cost is not None - assert generation.calculated_output_cost is not None - assert generation.calculated_total_cost is not None + assert generation.cost_details["input"] is not None + assert generation.cost_details["output"] is not None + assert generation.total_cost is not None assert generation.latency is not None @@ -100,8 +99,7 @@ def test_stream_completions_models(model_name): langfuse_client.flush() assert handler.runs == {} - api = get_api() - trace = api.trace.get(trace_id) + trace = wait_for_trace_snapshot(trace_id, min_observations=2) generationList = list(filter(lambda o: o.type == "GENERATION", trace.observations)) assert len(generationList) != 0 @@ -116,14 +114,14 @@ def test_stream_completions_models(model_name): assert generation.model_parameters.get("max_tokens") is not None assert generation.model_parameters.get("temperature") is not None assert generation.metadata["tags"] == tags - assert generation.usage.output is not None - assert generation.usage.total is not None + assert generation.usage_details["output"] is not None + assert generation.usage_details["total"] is not None assert generation.output is not None assert generation.input_price is not None assert generation.output_price is not None - assert generation.calculated_input_cost is not None - assert generation.calculated_output_cost is not None - assert generation.calculated_total_cost is not None + assert generation.cost_details["input"] is not None + assert generation.cost_details["output"] is not None + assert generation.total_cost is not None assert generation.latency is not None @@ -150,8 +148,7 @@ def test_invoke_chat_models(model_name): langfuse_client.flush() assert handler.runs == {} - api = get_api() - trace = api.trace.get(trace_id) + trace = wait_for_trace_snapshot(trace_id, min_observations=2) generationList = list(filter(lambda o: o.type == "GENERATION", trace.observations)) assert len(generationList) != 0 @@ -165,15 +162,15 @@ def test_invoke_chat_models(model_name): assert generation.model_parameters.get("max_completion_tokens") is not None assert generation.model_parameters.get("temperature") is not None assert generation.metadata["tags"] == tags - assert generation.usage.output is not None - assert generation.usage.total is not None + assert generation.usage_details["output"] is not None + assert generation.usage_details["total"] is not None assert generation.output["content"] is not None assert generation.output["role"] is not None assert generation.input_price is not None assert generation.output_price is not None - assert generation.calculated_input_cost is not None - assert generation.calculated_output_cost is not None - assert generation.calculated_total_cost is not None + assert generation.cost_details["input"] is not None + assert generation.cost_details["output"] is not None + assert generation.total_cost is not None assert generation.latency is not None @@ -201,8 +198,7 @@ def test_invoke_in_completions_models(model_name): langfuse_client.flush() assert handler.runs == {} - api = get_api() - trace = api.trace.get(trace_id) + trace = wait_for_trace_snapshot(trace_id, min_observations=2) generationList = list(filter(lambda o: o.type == "GENERATION", trace.observations)) assert len(generationList) != 0 @@ -216,14 +212,14 @@ def test_invoke_in_completions_models(model_name): assert generation.model_parameters.get("max_tokens") is not None assert generation.model_parameters.get("temperature") is not None assert generation.metadata["tags"] == tags - assert generation.usage.output is not None - assert generation.usage.total is not None + assert generation.usage_details["output"] is not None + assert generation.usage_details["total"] is not None assert test_phrase in generation.output assert generation.input_price is not None assert generation.output_price is not None - assert generation.calculated_input_cost is not None - assert generation.calculated_output_cost is not None - assert generation.calculated_total_cost is not None + assert generation.cost_details["input"] is not None + assert generation.cost_details["output"] is not None + assert generation.total_cost is not None assert generation.latency is not None @@ -251,8 +247,7 @@ def test_batch_in_completions_models(model_name): langfuse_client.flush() assert handler.runs == {} - api = get_api() - trace = api.trace.get(trace_id) + trace = wait_for_trace_snapshot(trace_id, min_observations=3) generationList = list(filter(lambda o: o.type == "GENERATION", trace.observations)) assert len(generationList) != 0 @@ -266,13 +261,13 @@ def test_batch_in_completions_models(model_name): assert generation.model_parameters.get("max_tokens") is not None assert generation.model_parameters.get("temperature") is not None assert generation.metadata["tags"] == tags - assert generation.usage.output is not None - assert generation.usage.total is not None + assert generation.usage_details["output"] is not None + assert generation.usage_details["total"] is not None assert generation.input_price is not None assert generation.output_price is not None - assert generation.calculated_input_cost is not None - assert generation.calculated_output_cost is not None - assert generation.calculated_total_cost is not None + assert generation.cost_details["input"] is not None + assert generation.cost_details["output"] is not None + assert generation.total_cost is not None assert generation.latency is not None @@ -300,8 +295,7 @@ def test_batch_in_chat_models(model_name): langfuse_client.flush() assert handler.runs == {} - api = get_api() - trace = api.trace.get(trace_id) + trace = wait_for_trace_snapshot(trace_id, min_observations=3) generationList = list(filter(lambda o: o.type == "GENERATION", trace.observations)) assert len(generationList) != 0 @@ -314,13 +308,13 @@ def test_batch_in_chat_models(model_name): assert generation.model_parameters.get("max_completion_tokens") is not None assert generation.model_parameters.get("temperature") is not None assert generation.metadata["tags"] == tags - assert generation.usage.output is not None - assert generation.usage.total is not None + assert generation.usage_details["output"] is not None + assert generation.usage_details["total"] is not None assert generation.input_price is not None assert generation.output_price is not None - assert generation.calculated_input_cost is not None - assert generation.calculated_output_cost is not None - assert generation.calculated_total_cost is not None + assert generation.cost_details["input"] is not None + assert generation.cost_details["output"] is not None + assert generation.total_cost is not None assert generation.latency is not None @@ -354,8 +348,7 @@ async def test_astream_chat_models(model_name): langfuse_client.flush() assert handler.runs == {} - api = get_api() - trace = api.trace.get(trace_id) + trace = wait_for_trace_snapshot(trace_id, min_observations=2) generationList = list(filter(lambda o: o.type == "GENERATION", trace.observations)) assert len(generationList) != 0 @@ -369,15 +362,15 @@ async def test_astream_chat_models(model_name): assert generation.model_parameters.get("max_completion_tokens") is not None assert generation.model_parameters.get("temperature") is not None assert generation.metadata["tags"] == tags - assert generation.usage.output is not None - assert generation.usage.total is not None + assert generation.usage_details["output"] is not None + assert generation.usage_details["total"] is not None assert generation.output["content"] is not None assert generation.output["role"] is not None assert generation.input_price is not None assert generation.output_price is not None - assert generation.calculated_input_cost is not None - assert generation.calculated_output_cost is not None - assert generation.calculated_total_cost is not None + assert generation.cost_details["input"] is not None + assert generation.cost_details["output"] is not None + assert generation.total_cost is not None assert generation.latency is not None @@ -411,8 +404,7 @@ async def test_astream_completions_models(model_name): langfuse_client.flush() assert handler.runs == {} - api = get_api() - trace = api.trace.get(trace_id) + trace = wait_for_trace_snapshot(trace_id, min_observations=2) generationList = list(filter(lambda o: o.type == "GENERATION", trace.observations)) assert len(generationList) != 0 @@ -427,14 +419,14 @@ async def test_astream_completions_models(model_name): assert generation.model_parameters.get("max_tokens") is not None assert generation.model_parameters.get("temperature") is not None assert generation.metadata["tags"] == tags - assert generation.usage.output is not None - assert generation.usage.total is not None + assert generation.usage_details["output"] is not None + assert generation.usage_details["total"] is not None assert test_phrase in generation.output assert generation.input_price is not None assert generation.output_price is not None - assert generation.calculated_input_cost is not None - assert generation.calculated_output_cost is not None - assert generation.calculated_total_cost is not None + assert generation.cost_details["input"] is not None + assert generation.cost_details["output"] is not None + assert generation.total_cost is not None assert generation.latency is not None @@ -463,8 +455,7 @@ async def test_ainvoke_chat_models(model_name): langfuse_client.flush() assert handler.runs == {} - api = get_api() - trace = api.trace.get(trace_id) + trace = wait_for_trace_snapshot(trace_id, min_observations=2) generationList = list(filter(lambda o: o.type == "GENERATION", trace.observations)) assert len(generationList) != 0 @@ -478,15 +469,15 @@ async def test_ainvoke_chat_models(model_name): assert generation.model_parameters.get("max_completion_tokens") is not None assert generation.model_parameters.get("temperature") is not None assert generation.metadata["tags"] == tags - assert generation.usage.output is not None - assert generation.usage.total is not None + assert generation.usage_details["output"] is not None + assert generation.usage_details["total"] is not None assert generation.output["content"] is not None assert generation.output["role"] is not None assert generation.input_price is not None assert generation.output_price is not None - assert generation.calculated_input_cost is not None - assert generation.calculated_output_cost is not None - assert generation.calculated_total_cost is not None + assert generation.cost_details["input"] is not None + assert generation.cost_details["output"] is not None + assert generation.total_cost is not None assert generation.latency is not None @@ -514,8 +505,7 @@ async def test_ainvoke_in_completions_models(model_name): langfuse_client.flush() assert handler.runs == {} - api = get_api() - trace = api.trace.get(trace_id) + trace = wait_for_trace_snapshot(trace_id, min_observations=2) generationList = list(filter(lambda o: o.type == "GENERATION", trace.observations)) assert len(generationList) != 0 @@ -529,14 +519,14 @@ async def test_ainvoke_in_completions_models(model_name): assert generation.model_parameters.get("max_tokens") is not None assert generation.model_parameters.get("temperature") is not None assert generation.metadata["tags"] == tags - assert generation.usage.output is not None - assert generation.usage.total is not None + assert generation.usage_details["output"] is not None + assert generation.usage_details["total"] is not None assert test_phrase in generation.output assert generation.input_price is not None assert generation.output_price is not None - assert generation.calculated_input_cost is not None - assert generation.calculated_output_cost is not None - assert generation.calculated_total_cost is not None + assert generation.cost_details["input"] is not None + assert generation.cost_details["output"] is not None + assert generation.total_cost is not None assert generation.latency is not None @@ -571,8 +561,7 @@ def test_chains_batch_in_chat_models(model_name): langfuse_client.flush() assert handler.runs == {} - api = get_api() - trace = api.trace.get(trace_id) + trace = wait_for_trace_snapshot(trace_id, min_observations=9) generationList = list(filter(lambda o: o.type == "GENERATION", trace.observations)) assert len(generationList) != 0 @@ -585,13 +574,13 @@ def test_chains_batch_in_chat_models(model_name): assert generation.model_parameters.get("max_completion_tokens") is not None assert generation.model_parameters.get("temperature") is not None assert all(x in generation.metadata["tags"] for x in tags) - assert generation.usage.output is not None - assert generation.usage.total is not None + assert generation.usage_details["output"] is not None + assert generation.usage_details["total"] is not None assert generation.input_price is not None assert generation.output_price is not None - assert generation.calculated_input_cost is not None - assert generation.calculated_output_cost is not None - assert generation.calculated_total_cost is not None + assert generation.cost_details["input"] is not None + assert generation.cost_details["output"] is not None + assert generation.total_cost is not None assert generation.latency is not None @@ -622,8 +611,7 @@ def test_chains_batch_in_completions_models(model_name): langfuse_client.flush() assert handler.runs == {} - api = get_api() - trace = api.trace.get(trace_id) + trace = wait_for_trace_snapshot(trace_id, min_observations=9) generationList = list(filter(lambda o: o.type == "GENERATION", trace.observations)) assert len(generationList) != 0 @@ -636,13 +624,13 @@ def test_chains_batch_in_completions_models(model_name): assert generation.model_parameters.get("max_tokens") is not None assert generation.model_parameters.get("temperature") is not None assert all(x in generation.metadata["tags"] for x in tags) - assert generation.usage.output is not None - assert generation.usage.total is not None + assert generation.usage_details["output"] is not None + assert generation.usage_details["total"] is not None assert generation.input_price is not None assert generation.output_price is not None - assert generation.calculated_input_cost is not None - assert generation.calculated_output_cost is not None - assert generation.calculated_total_cost is not None + assert generation.cost_details["input"] is not None + assert generation.cost_details["output"] is not None + assert generation.total_cost is not None assert generation.latency is not None @@ -675,8 +663,7 @@ async def test_chains_abatch_in_chat_models(model_name): langfuse_client.flush() assert handler.runs == {} - api = get_api() - trace = api.trace.get(trace_id) + trace = wait_for_trace_snapshot(trace_id, min_observations=9) generationList = list(filter(lambda o: o.type == "GENERATION", trace.observations)) assert len(generationList) != 0 @@ -689,13 +676,13 @@ async def test_chains_abatch_in_chat_models(model_name): assert generation.model_parameters.get("max_completion_tokens") is not None assert generation.model_parameters.get("temperature") is not None assert all(x in generation.metadata["tags"] for x in tags) - assert generation.usage.output is not None - assert generation.usage.total is not None + assert generation.usage_details["output"] is not None + assert generation.usage_details["total"] is not None assert generation.input_price is not None assert generation.output_price is not None - assert generation.calculated_input_cost is not None - assert generation.calculated_output_cost is not None - assert generation.calculated_total_cost is not None + assert generation.cost_details["input"] is not None + assert generation.cost_details["output"] is not None + assert generation.total_cost is not None assert generation.latency is not None @@ -725,8 +712,7 @@ async def test_chains_abatch_in_completions_models(model_name): langfuse_client.flush() assert handler.runs == {} - api = get_api() - trace = api.trace.get(trace_id) + trace = wait_for_trace_snapshot(trace_id, min_observations=9) generationList = list(filter(lambda o: o.type == "GENERATION", trace.observations)) assert len(generationList) != 0 assert len(trace.observations) == 9 @@ -738,13 +724,13 @@ async def test_chains_abatch_in_completions_models(model_name): assert generation.model_parameters.get("max_tokens") is not None assert generation.model_parameters.get("temperature") is not None assert all(x in generation.metadata["tags"] for x in tags) - assert generation.usage.output is not None - assert generation.usage.total is not None + assert generation.usage_details["output"] is not None + assert generation.usage_details["total"] is not None assert generation.input_price is not None assert generation.output_price is not None - assert generation.calculated_input_cost is not None - assert generation.calculated_output_cost is not None - assert generation.calculated_total_cost is not None + assert generation.cost_details["input"] is not None + assert generation.cost_details["output"] is not None + assert generation.total_cost is not None assert generation.latency is not None @@ -778,8 +764,7 @@ async def test_chains_ainvoke_chat_models(model_name): langfuse_client.flush() assert handler.runs == {} - api = get_api() - trace = api.trace.get(trace_id) + trace = wait_for_trace_snapshot(trace_id, min_observations=5) generationList = list(filter(lambda o: o.type == "GENERATION", trace.observations)) assert len(generationList) != 0 @@ -792,15 +777,15 @@ async def test_chains_ainvoke_chat_models(model_name): assert generation.model_parameters.get("max_completion_tokens") is not None assert generation.model_parameters.get("temperature") is not None assert all(x in generation.metadata["tags"] for x in tags) - assert generation.usage.output is not None - assert generation.usage.total is not None + assert generation.usage_details["output"] is not None + assert generation.usage_details["total"] is not None assert generation.output["content"] is not None assert generation.output["role"] is not None assert generation.input_price is not None assert generation.output_price is not None - assert generation.calculated_input_cost is not None - assert generation.calculated_output_cost is not None - assert generation.calculated_total_cost is not None + assert generation.cost_details["input"] is not None + assert generation.cost_details["output"] is not None + assert generation.total_cost is not None assert generation.latency is not None @@ -834,8 +819,7 @@ async def test_chains_ainvoke_completions_models(model_name): langfuse_client.flush() assert handler.runs == {} - api = get_api() - trace = api.trace.get(trace_id) + trace = wait_for_trace_snapshot(trace_id, min_observations=5) generationList = list(filter(lambda o: o.type == "GENERATION", trace.observations)) assert len(generationList) != 0 @@ -848,13 +832,13 @@ async def test_chains_ainvoke_completions_models(model_name): assert generation.model_parameters.get("max_tokens") is not None assert generation.model_parameters.get("temperature") is not None assert all(x in generation.metadata["tags"] for x in tags) - assert generation.usage.output is not None - assert generation.usage.total is not None + assert generation.usage_details["output"] is not None + assert generation.usage_details["total"] is not None assert generation.input_price is not None assert generation.output_price is not None - assert generation.calculated_input_cost is not None - assert generation.calculated_output_cost is not None - assert generation.calculated_total_cost is not None + assert generation.cost_details["input"] is not None + assert generation.cost_details["output"] is not None + assert generation.total_cost is not None assert generation.latency is not None @@ -894,8 +878,7 @@ async def test_chains_astream_chat_models(model_name): langfuse_client.flush() assert handler.runs == {} - api = get_api() - trace = api.trace.get(trace_id) + trace = wait_for_trace_snapshot(trace_id, min_observations=5) generationList = list(filter(lambda o: o.type == "GENERATION", trace.observations)) assert len(generationList) != 0 @@ -910,15 +893,15 @@ async def test_chains_astream_chat_models(model_name): assert generation.model_parameters.get("max_completion_tokens") is not None assert generation.model_parameters.get("temperature") is not None assert all(x in generation.metadata["tags"] for x in tags) - assert generation.usage.output is not None - assert generation.usage.total is not None + assert generation.usage_details["output"] is not None + assert generation.usage_details["total"] is not None assert generation.output["content"] is not None assert generation.output["role"] is not None assert generation.input_price is not None assert generation.output_price is not None - assert generation.calculated_input_cost is not None - assert generation.calculated_output_cost is not None - assert generation.calculated_total_cost is not None + assert generation.cost_details["input"] is not None + assert generation.cost_details["output"] is not None + assert generation.total_cost is not None assert generation.latency is not None @@ -956,8 +939,7 @@ async def test_chains_astream_completions_models(model_name): langfuse_client.flush() assert handler.runs == {} - api = get_api() - trace = api.trace.get(trace_id) + trace = wait_for_trace_snapshot(trace_id, min_observations=5) generationList = list(filter(lambda o: o.type == "GENERATION", trace.observations)) assert len(generationList) != 0 @@ -972,11 +954,11 @@ async def test_chains_astream_completions_models(model_name): assert generation.model_parameters.get("max_tokens") is not None assert generation.model_parameters.get("temperature") is not None assert all(x in generation.metadata["tags"] for x in tags) - assert generation.usage.output is not None - assert generation.usage.total is not None + assert generation.usage_details["output"] is not None + assert generation.usage_details["total"] is not None assert generation.input_price is not None assert generation.output_price is not None - assert generation.calculated_input_cost is not None - assert generation.calculated_output_cost is not None - assert generation.calculated_total_cost is not None + assert generation.cost_details["input"] is not None + assert generation.cost_details["output"] is not None + assert generation.total_cost is not None assert generation.latency is not None diff --git a/tests/live_provider/test_openai.py b/tests/live_provider/test_openai.py index 43a07be23..18dc61b81 100644 --- a/tests/live_provider/test_openai.py +++ b/tests/live_provider/test_openai.py @@ -6,7 +6,11 @@ from pydantic import BaseModel from langfuse._client.client import Langfuse -from tests.support.utils import create_uuid, encode_file_to_base64, get_api +from tests.support.utils import ( + create_uuid, + encode_file_to_base64, + wait_for_observations, +) langfuse: Langfuse | None = None @@ -58,27 +62,25 @@ def test_openai_chat_completion(openai): sleep(1) - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) + generation = wait_for_observations(name=generation_name, type="GENERATION") - assert len(generation.data) != 0 - assert generation.data[0].name == generation_name - assert generation.data[0].metadata["someKey"] == "someResponse" + assert len(generation) != 0 + assert generation[0].name == generation_name + assert generation[0].metadata["someKey"] == "someResponse" assert len(completion.choices) != 0 - assert generation.data[0].input == [ + assert generation[0].input == [ { "content": "You are an expert mathematician", "role": "assistant", }, {"content": "1 + 1 = ", "role": "user"}, ] - assert generation.data[0].type == "GENERATION" - assert "gpt-3.5-turbo-0125" in generation.data[0].model - assert generation.data[0].start_time is not None - assert generation.data[0].end_time is not None - assert generation.data[0].start_time < generation.data[0].end_time - assert generation.data[0].model_parameters == { + assert generation[0].type == "GENERATION" + assert "gpt-3.5-turbo-0125" in generation[0].model + assert generation[0].start_time is not None + assert generation[0].end_time is not None + assert generation[0].start_time < generation[0].end_time + assert generation[0].model_parameters == { "service_tier": "default", "temperature": 0, "top_p": 1, @@ -86,11 +88,11 @@ def test_openai_chat_completion(openai): "max_tokens": "Infinity", "presence_penalty": 0, } - assert generation.data[0].usage.input is not None - assert generation.data[0].usage.output is not None - assert generation.data[0].usage.total is not None - assert "2" in generation.data[0].output["content"] - assert generation.data[0].output["role"] == "assistant" + assert generation[0].usage_details["input"] is not None + assert generation[0].usage_details["output"] is not None + assert generation[0].usage_details["total"] is not None + assert "2" in generation[0].output["content"] + assert generation[0].output["role"] == "assistant" def test_openai_chat_completion_stream(openai): @@ -116,21 +118,19 @@ def test_openai_chat_completion_stream(openai): langfuse.flush() sleep(3) - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) + generation = wait_for_observations(name=generation_name, type="GENERATION") + + assert len(generation) != 0 + assert generation[0].name == generation_name + assert generation[0].metadata["someKey"] == "someResponse" - assert len(generation.data) != 0 - assert generation.data[0].name == generation_name - assert generation.data[0].metadata["someKey"] == "someResponse" - - assert generation.data[0].input == [{"content": "1 + 1 = ", "role": "user"}] - assert generation.data[0].type == "GENERATION" - assert "gpt-3.5-turbo-0125" in generation.data[0].model - assert generation.data[0].start_time is not None - assert generation.data[0].end_time is not None - assert generation.data[0].start_time < generation.data[0].end_time - assert generation.data[0].model_parameters == { + assert generation[0].input == [{"content": "1 + 1 = ", "role": "user"}] + assert generation[0].type == "GENERATION" + assert "gpt-3.5-turbo-0125" in generation[0].model + assert generation[0].start_time is not None + assert generation[0].end_time is not None + assert generation[0].start_time < generation[0].end_time + assert generation[0].model_parameters == { "service_tier": "default", "temperature": 0, "top_p": 1, @@ -138,16 +138,16 @@ def test_openai_chat_completion_stream(openai): "max_tokens": "Infinity", "presence_penalty": 0, } - assert generation.data[0].usage.input is not None - assert generation.data[0].usage.output is not None - assert generation.data[0].usage.total is not None - assert generation.data[0].output == 2 - assert generation.data[0].completion_start_time is not None + assert generation[0].usage_details["input"] is not None + assert generation[0].usage_details["output"] is not None + assert generation[0].usage_details["total"] is not None + assert generation[0].output == "2" + assert generation[0].completion_start_time is not None # Completion start time for time-to-first-token - assert generation.data[0].completion_start_time is not None - assert generation.data[0].completion_start_time >= generation.data[0].start_time - assert generation.data[0].completion_start_time <= generation.data[0].end_time + assert generation[0].completion_start_time is not None + assert generation[0].completion_start_time >= generation[0].start_time + assert generation[0].completion_start_time <= generation[0].end_time def test_openai_chat_completion_stream_with_next_iteration(openai): @@ -177,21 +177,19 @@ def test_openai_chat_completion_stream_with_next_iteration(openai): langfuse.flush() - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) + generation = wait_for_observations(name=generation_name, type="GENERATION") + + assert len(generation) != 0 + assert generation[0].name == generation_name + assert generation[0].metadata["someKey"] == "someResponse" - assert len(generation.data) != 0 - assert generation.data[0].name == generation_name - assert generation.data[0].metadata["someKey"] == "someResponse" - - assert generation.data[0].input == [{"content": "1 + 1 = ", "role": "user"}] - assert generation.data[0].type == "GENERATION" - assert generation.data[0].model == "gpt-3.5-turbo-0125" - assert generation.data[0].start_time is not None - assert generation.data[0].end_time is not None - assert generation.data[0].start_time < generation.data[0].end_time - assert generation.data[0].model_parameters == { + assert generation[0].input == [{"content": "1 + 1 = ", "role": "user"}] + assert generation[0].type == "GENERATION" + assert generation[0].model == "gpt-3.5-turbo-0125" + assert generation[0].start_time is not None + assert generation[0].end_time is not None + assert generation[0].start_time < generation[0].end_time + assert generation[0].model_parameters == { "service_tier": "default", "temperature": 0, "top_p": 1, @@ -199,16 +197,16 @@ def test_openai_chat_completion_stream_with_next_iteration(openai): "max_tokens": "Infinity", "presence_penalty": 0, } - assert generation.data[0].usage.input is not None - assert generation.data[0].usage.output is not None - assert generation.data[0].usage.total is not None - assert generation.data[0].output == 2 - assert generation.data[0].completion_start_time is not None + assert generation[0].usage_details["input"] is not None + assert generation[0].usage_details["output"] is not None + assert generation[0].usage_details["total"] is not None + assert generation[0].output == "2" + assert generation[0].completion_start_time is not None # Completion start time for time-to-first-token - assert generation.data[0].completion_start_time is not None - assert generation.data[0].completion_start_time >= generation.data[0].start_time - assert generation.data[0].completion_start_time <= generation.data[0].end_time + assert generation[0].completion_start_time is not None + assert generation[0].completion_start_time >= generation[0].start_time + assert generation[0].completion_start_time <= generation[0].end_time def test_openai_chat_completion_stream_fail(openai): @@ -227,33 +225,28 @@ def test_openai_chat_completion_stream_fail(openai): langfuse.flush() - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) + generation = wait_for_observations(name=generation_name, type="GENERATION") - assert len(generation.data) != 0 - assert generation.data[0].name == generation_name - assert generation.data[0].metadata["someKey"] == "someResponse" - - assert generation.data[0].input == [{"content": "1 + 1 = ", "role": "user"}] - assert generation.data[0].type == "GENERATION" - assert generation.data[0].model == "fake" - assert generation.data[0].start_time is not None - assert generation.data[0].end_time is not None - assert generation.data[0].start_time < generation.data[0].end_time - assert generation.data[0].model_parameters == { + assert len(generation) != 0 + assert generation[0].name == generation_name + assert generation[0].metadata["someKey"] == "someResponse" + + assert generation[0].input == [{"content": "1 + 1 = ", "role": "user"}] + assert generation[0].type == "GENERATION" + assert generation[0].model == "fake" + assert generation[0].start_time is not None + assert generation[0].end_time is not None + assert generation[0].start_time < generation[0].end_time + assert generation[0].model_parameters == { "temperature": 0, "top_p": 1, "frequency_penalty": 0, "max_tokens": "Infinity", "presence_penalty": 0, } - assert generation.data[0].usage.input is not None - assert generation.data[0].usage.output is not None - assert generation.data[0].usage.total is not None - assert generation.data[0].level == "ERROR" - assert generation.data[0].status_message is not None - assert generation.data[0].output is None + assert generation[0].level == "ERROR" + assert generation[0].status_message is not None + assert generation[0].output is None openai.api_key = os.environ["OPENAI_API_KEY"] @@ -277,13 +270,11 @@ def test_openai_chat_completion_with_langfuse_prompt(openai): langfuse.flush() - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) + generation = wait_for_observations(name=generation_name, type="GENERATION") - assert len(generation.data) != 0 - assert generation.data[0].name == generation_name - assert isinstance(generation.data[0].prompt_id, str) + assert len(generation) != 0 + assert generation[0].name == generation_name + assert isinstance(generation[0].prompt_id, str) def test_openai_chat_completion_fail(openai): @@ -300,29 +291,27 @@ def test_openai_chat_completion_fail(openai): langfuse.flush() - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) - - assert len(generation.data) != 0 - assert generation.data[0].name == generation_name - assert generation.data[0].metadata["someKey"] == "someResponse" - assert generation.data[0].input == [{"content": "1 + 1 = ", "role": "user"}] - assert generation.data[0].type == "GENERATION" - assert generation.data[0].model == "fake" - assert generation.data[0].level == "ERROR" - assert generation.data[0].start_time is not None - assert generation.data[0].end_time is not None - assert generation.data[0].status_message is not None - assert generation.data[0].start_time < generation.data[0].end_time - assert generation.data[0].model_parameters == { + generation = wait_for_observations(name=generation_name, type="GENERATION") + + assert len(generation) != 0 + assert generation[0].name == generation_name + assert generation[0].metadata["someKey"] == "someResponse" + assert generation[0].input == [{"content": "1 + 1 = ", "role": "user"}] + assert generation[0].type == "GENERATION" + assert generation[0].model == "fake" + assert generation[0].level == "ERROR" + assert generation[0].start_time is not None + assert generation[0].end_time is not None + assert generation[0].status_message is not None + assert generation[0].start_time < generation[0].end_time + assert generation[0].model_parameters == { "temperature": 0, "top_p": 1, "frequency_penalty": 0, "max_tokens": "Infinity", "presence_penalty": 0, } - assert generation.data[0].output is None + assert generation[0].output is None openai.api_key = os.environ["OPENAI_API_KEY"] @@ -360,25 +349,21 @@ def test_openai_chat_completion_two_calls(openai): langfuse.flush() - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) + generation = wait_for_observations(name=generation_name, type="GENERATION") - assert len(generation.data) != 0 - assert generation.data[0].name == generation_name + assert len(generation) != 0 + assert generation[0].name == generation_name assert len(completion.choices) != 0 - assert generation.data[0].input == [{"content": "1 + 1 = ", "role": "user"}] + assert generation[0].input == [{"content": "1 + 1 = ", "role": "user"}] - generation_2 = get_api().legacy.observations_v1.get_many( - name=generation_name_2, type="GENERATION" - ) + generation_2 = wait_for_observations(name=generation_name_2, type="GENERATION") - assert len(generation_2.data) != 0 - assert generation_2.data[0].name == generation_name_2 + assert len(generation_2) != 0 + assert generation_2[0].name == generation_name_2 assert len(completion_2.choices) != 0 - assert generation_2.data[0].input == [{"content": "2 + 2 = ", "role": "user"}] + assert generation_2[0].input == [{"content": "2 + 2 = ", "role": "user"}] def test_openai_chat_completion_with_seed(openai): @@ -394,11 +379,9 @@ def test_openai_chat_completion_with_seed(openai): langfuse.flush() - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) + generation = wait_for_observations(name=generation_name, type="GENERATION") - assert generation.data[0].model_parameters == { + assert generation[0].model_parameters == { "service_tier": "default", "temperature": 0, "top_p": 1, @@ -423,32 +406,30 @@ def test_openai_completion(openai): langfuse.flush() - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) + generation = wait_for_observations(name=generation_name, type="GENERATION") - assert len(generation.data) != 0 - assert generation.data[0].name == generation_name - assert generation.data[0].metadata["someKey"] == "someResponse" + assert len(generation) != 0 + assert generation[0].name == generation_name + assert generation[0].metadata["someKey"] == "someResponse" assert len(completion.choices) != 0 - assert completion.choices[0].text == generation.data[0].output - assert generation.data[0].input == "1 + 1 = " - assert generation.data[0].type == "GENERATION" - assert "gpt-3.5-turbo-instruct" in generation.data[0].model - assert generation.data[0].start_time is not None - assert generation.data[0].end_time is not None - assert generation.data[0].start_time < generation.data[0].end_time - assert generation.data[0].model_parameters == { + assert completion.choices[0].text == generation[0].output + assert generation[0].input == "1 + 1 = " + assert generation[0].type == "GENERATION" + assert "gpt-3.5-turbo-instruct" in generation[0].model + assert generation[0].start_time is not None + assert generation[0].end_time is not None + assert generation[0].start_time < generation[0].end_time + assert generation[0].model_parameters == { "temperature": 0, "top_p": 1, "frequency_penalty": 0, "max_tokens": "Infinity", "presence_penalty": 0, } - assert generation.data[0].usage.input is not None - assert generation.data[0].usage.output is not None - assert generation.data[0].usage.total is not None - assert generation.data[0].output == "2\n\n1 + 2 = 3\n\n2 + 3 = " + assert generation[0].usage_details["input"] is not None + assert generation[0].usage_details["output"] is not None + assert generation[0].usage_details["total"] is not None + assert generation[0].output == "2\n\n1 + 2 = 3\n\n2 + 3 = " @requires_legacy_completion_model @@ -472,37 +453,35 @@ def test_openai_completion_stream(openai): assert len(content) > 0 - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) + generation = wait_for_observations(name=generation_name, type="GENERATION") - assert len(generation.data) != 0 - assert generation.data[0].name == generation_name - assert generation.data[0].metadata["someKey"] == "someResponse" - - assert generation.data[0].input == "1 + 1 = " - assert generation.data[0].type == "GENERATION" - assert "gpt-3.5-turbo-instruct" in generation.data[0].model - assert generation.data[0].start_time is not None - assert generation.data[0].end_time is not None - assert generation.data[0].start_time < generation.data[0].end_time - assert generation.data[0].model_parameters == { + assert len(generation) != 0 + assert generation[0].name == generation_name + assert generation[0].metadata["someKey"] == "someResponse" + + assert generation[0].input == "1 + 1 = " + assert generation[0].type == "GENERATION" + assert "gpt-3.5-turbo-instruct" in generation[0].model + assert generation[0].start_time is not None + assert generation[0].end_time is not None + assert generation[0].start_time < generation[0].end_time + assert generation[0].model_parameters == { "temperature": 0, "top_p": 1, "frequency_penalty": 0, "max_tokens": "Infinity", "presence_penalty": 0, } - assert generation.data[0].usage.input is not None - assert generation.data[0].usage.output is not None - assert generation.data[0].usage.total is not None - assert generation.data[0].output == "2\n\n1 + 2 = 3\n\n2 + 3 = " - assert generation.data[0].completion_start_time is not None + assert generation[0].usage_details["input"] is not None + assert generation[0].usage_details["output"] is not None + assert generation[0].usage_details["total"] is not None + assert generation[0].output == "2\n\n1 + 2 = 3\n\n2 + 3 = " + assert generation[0].completion_start_time is not None # Completion start time for time-to-first-token - assert generation.data[0].completion_start_time is not None - assert generation.data[0].completion_start_time >= generation.data[0].start_time - assert generation.data[0].completion_start_time <= generation.data[0].end_time + assert generation[0].completion_start_time is not None + assert generation[0].completion_start_time >= generation[0].start_time + assert generation[0].completion_start_time <= generation[0].end_time def test_openai_completion_fail(openai): @@ -521,29 +500,27 @@ def test_openai_completion_fail(openai): langfuse.flush() - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) - - assert len(generation.data) != 0 - assert generation.data[0].name == generation_name - assert generation.data[0].metadata["someKey"] == "someResponse" - assert generation.data[0].input == "1 + 1 = " - assert generation.data[0].type == "GENERATION" - assert generation.data[0].model == "fake" - assert generation.data[0].level == "ERROR" - assert generation.data[0].start_time is not None - assert generation.data[0].end_time is not None - assert generation.data[0].status_message is not None - assert generation.data[0].start_time < generation.data[0].end_time - assert generation.data[0].model_parameters == { + generation = wait_for_observations(name=generation_name, type="GENERATION") + + assert len(generation) != 0 + assert generation[0].name == generation_name + assert generation[0].metadata["someKey"] == "someResponse" + assert generation[0].input == "1 + 1 = " + assert generation[0].type == "GENERATION" + assert generation[0].model == "fake" + assert generation[0].level == "ERROR" + assert generation[0].start_time is not None + assert generation[0].end_time is not None + assert generation[0].status_message is not None + assert generation[0].start_time < generation[0].end_time + assert generation[0].model_parameters == { "temperature": 0, "top_p": 1, "frequency_penalty": 0, "max_tokens": "Infinity", "presence_penalty": 0, } - assert generation.data[0].output is None + assert generation[0].output is None openai.api_key = os.environ["OPENAI_API_KEY"] @@ -564,33 +541,28 @@ def test_openai_completion_stream_fail(openai): langfuse.flush() - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) + generation = wait_for_observations(name=generation_name, type="GENERATION") + + assert len(generation) != 0 + assert generation[0].name == generation_name + assert generation[0].metadata["someKey"] == "someResponse" - assert len(generation.data) != 0 - assert generation.data[0].name == generation_name - assert generation.data[0].metadata["someKey"] == "someResponse" - - assert generation.data[0].input == "1 + 1 = " - assert generation.data[0].type == "GENERATION" - assert generation.data[0].model == "gpt-3.5-turbo" - assert generation.data[0].start_time is not None - assert generation.data[0].end_time is not None - assert generation.data[0].start_time < generation.data[0].end_time - assert generation.data[0].model_parameters == { + assert generation[0].input == "1 + 1 = " + assert generation[0].type == "GENERATION" + assert generation[0].model == "gpt-3.5-turbo" + assert generation[0].start_time is not None + assert generation[0].end_time is not None + assert generation[0].start_time < generation[0].end_time + assert generation[0].model_parameters == { "temperature": 0, "top_p": 1, "frequency_penalty": 0, "max_tokens": "Infinity", "presence_penalty": 0, } - assert generation.data[0].usage.input is not None - assert generation.data[0].usage.output is not None - assert generation.data[0].usage.total is not None - assert generation.data[0].level == "ERROR" - assert generation.data[0].status_message is not None - assert generation.data[0].output is None + assert generation[0].level == "ERROR" + assert generation[0].status_message is not None + assert generation[0].output is None openai.api_key = os.environ["OPENAI_API_KEY"] @@ -614,13 +586,11 @@ def test_openai_completion_with_langfuse_prompt(openai): langfuse.flush() - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) + generation = wait_for_observations(name=generation_name, type="GENERATION") - assert len(generation.data) != 0 - assert generation.data[0].name == generation_name - assert isinstance(generation.data[0].prompt_id, str) + assert len(generation) != 0 + assert generation[0].name == generation_name + assert isinstance(generation[0].prompt_id, str) def test_fails_wrong_name(openai): @@ -656,21 +626,19 @@ async def test_async_chat(openai): langfuse.flush() - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) + generation = wait_for_observations(name=generation_name, type="GENERATION") - assert len(generation.data) != 0 - assert generation.data[0].name == generation_name + assert len(generation) != 0 + assert generation[0].name == generation_name assert len(completion.choices) != 0 - assert generation.data[0].input == [{"content": "1 + 1 = ", "role": "user"}] - assert generation.data[0].type == "GENERATION" - assert generation.data[0].model == "gpt-3.5-turbo-0125" - assert generation.data[0].start_time is not None - assert generation.data[0].end_time is not None - assert generation.data[0].start_time < generation.data[0].end_time - assert generation.data[0].model_parameters == { + assert generation[0].input == [{"content": "1 + 1 = ", "role": "user"}] + assert generation[0].type == "GENERATION" + assert generation[0].model == "gpt-3.5-turbo-0125" + assert generation[0].start_time is not None + assert generation[0].end_time is not None + assert generation[0].start_time < generation[0].end_time + assert generation[0].model_parameters == { "service_tier": "default", "temperature": 1, "top_p": 1, @@ -678,11 +646,11 @@ async def test_async_chat(openai): "max_tokens": "Infinity", "presence_penalty": 0, } - assert generation.data[0].usage.input is not None - assert generation.data[0].usage.output is not None - assert generation.data[0].usage.total is not None - assert "2" in generation.data[0].output["content"] - assert generation.data[0].output["role"] == "assistant" + assert generation[0].usage_details["input"] is not None + assert generation[0].usage_details["output"] is not None + assert generation[0].usage_details["total"] is not None + assert "2" in generation[0].output["content"] + assert generation[0].output["role"] == "assistant" @pytest.mark.asyncio @@ -703,19 +671,17 @@ async def test_async_chat_stream(openai): langfuse.flush() - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) - - assert len(generation.data) != 0 - assert generation.data[0].name == generation_name - assert generation.data[0].input == [{"content": "1 + 1 = ", "role": "user"}] - assert generation.data[0].type == "GENERATION" - assert generation.data[0].model == "gpt-3.5-turbo-0125" - assert generation.data[0].start_time is not None - assert generation.data[0].end_time is not None - assert generation.data[0].start_time < generation.data[0].end_time - assert generation.data[0].model_parameters == { + generation = wait_for_observations(name=generation_name, type="GENERATION") + + assert len(generation) != 0 + assert generation[0].name == generation_name + assert generation[0].input == [{"content": "1 + 1 = ", "role": "user"}] + assert generation[0].type == "GENERATION" + assert generation[0].model == "gpt-3.5-turbo-0125" + assert generation[0].start_time is not None + assert generation[0].end_time is not None + assert generation[0].start_time < generation[0].end_time + assert generation[0].model_parameters == { "service_tier": "default", "temperature": 1, "top_p": 1, @@ -723,15 +689,15 @@ async def test_async_chat_stream(openai): "max_tokens": "Infinity", "presence_penalty": 0, } - assert generation.data[0].usage.input is not None - assert generation.data[0].usage.output is not None - assert generation.data[0].usage.total is not None - assert "2" in str(generation.data[0].output) + assert generation[0].usage_details["input"] is not None + assert generation[0].usage_details["output"] is not None + assert generation[0].usage_details["total"] is not None + assert "2" in str(generation[0].output) # Completion start time for time-to-first-token - assert generation.data[0].completion_start_time is not None - assert generation.data[0].completion_start_time >= generation.data[0].start_time - assert generation.data[0].completion_start_time <= generation.data[0].end_time + assert generation[0].completion_start_time is not None + assert generation[0].completion_start_time >= generation[0].start_time + assert generation[0].completion_start_time <= generation[0].end_time @pytest.mark.asyncio @@ -762,21 +728,19 @@ async def test_async_chat_stream_with_anext(openai): print(result) - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) + generation = wait_for_observations(name=generation_name, type="GENERATION") - assert len(generation.data) != 0 - assert generation.data[0].name == generation_name - assert generation.data[0].input == [ + assert len(generation) != 0 + assert generation[0].name == generation_name + assert generation[0].input == [ {"content": "Give me a one-liner joke", "role": "user"} ] - assert generation.data[0].type == "GENERATION" - assert generation.data[0].model == "gpt-3.5-turbo-0125" - assert generation.data[0].start_time is not None - assert generation.data[0].end_time is not None - assert generation.data[0].start_time < generation.data[0].end_time - assert generation.data[0].model_parameters == { + assert generation[0].type == "GENERATION" + assert generation[0].model == "gpt-3.5-turbo-0125" + assert generation[0].start_time is not None + assert generation[0].end_time is not None + assert generation[0].start_time < generation[0].end_time + assert generation[0].model_parameters == { "service_tier": "default", "temperature": 1, "top_p": 1, @@ -784,14 +748,14 @@ async def test_async_chat_stream_with_anext(openai): "max_tokens": "Infinity", "presence_penalty": 0, } - assert generation.data[0].usage.input is not None - assert generation.data[0].usage.output is not None - assert generation.data[0].usage.total is not None + assert generation[0].usage_details["input"] is not None + assert generation[0].usage_details["output"] is not None + assert generation[0].usage_details["total"] is not None # Completion start time for time-to-first-token - assert generation.data[0].completion_start_time is not None - assert generation.data[0].completion_start_time >= generation.data[0].start_time - assert generation.data[0].completion_start_time <= generation.data[0].end_time + assert generation[0].completion_start_time is not None + assert generation[0].completion_start_time >= generation[0].start_time + assert generation[0].completion_start_time <= generation[0].end_time def test_openai_function_call(openai): @@ -825,14 +789,12 @@ class StepByStepAIResponse(BaseModel): langfuse.flush() - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) + generation = wait_for_observations(name=generation_name, type="GENERATION") - assert len(generation.data) != 0 - assert generation.data[0].name == generation_name - assert generation.data[0].output is not None - assert "function_call" in generation.data[0].output + assert len(generation) != 0 + assert generation[0].name == generation_name + assert generation[0].output is not None + assert "function_call" in generation[0].output assert output["title"] is not None @@ -869,14 +831,12 @@ class StepByStepAIResponse(BaseModel): langfuse.flush() - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) + generation = wait_for_observations(name=generation_name, type="GENERATION") - assert len(generation.data) != 0 - assert generation.data[0].name == generation_name - assert generation.data[0].output is not None - assert "function_call" in generation.data[0].output + assert len(generation) != 0 + assert generation[0].name == generation_name + assert generation[0].output is not None + assert "function_call" in generation[0].output def test_openai_tool_call(openai): @@ -913,21 +873,17 @@ def test_openai_tool_call(openai): langfuse.flush() - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) + generation = wait_for_observations(name=generation_name, type="GENERATION") - assert len(generation.data) != 0 - assert generation.data[0].name == generation_name + assert len(generation) != 0 + assert generation[0].name == generation_name assert ( - generation.data[0].output["tool_calls"][0]["function"]["name"] + generation[0].output["tool_calls"][0]["function"]["name"] == "get_current_weather" ) - assert ( - generation.data[0].output["tool_calls"][0]["function"]["arguments"] is not None - ) - assert generation.data[0].input["tools"] == tools - assert generation.data[0].input["messages"] == messages + assert generation[0].output["tool_calls"][0]["function"]["arguments"] is not None + assert generation[0].input["tools"] == tools + assert generation[0].input["messages"] == messages def test_openai_tool_call_streamed(openai): @@ -969,22 +925,18 @@ def test_openai_tool_call_streamed(openai): langfuse.flush() - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) + generation = wait_for_observations(name=generation_name, type="GENERATION") - assert len(generation.data) != 0 - assert generation.data[0].name == generation_name + assert len(generation) != 0 + assert generation[0].name == generation_name assert ( - generation.data[0].output["tool_calls"][0]["function"]["name"] + generation[0].output["tool_calls"][0]["function"]["name"] == "get_current_weather" ) - assert ( - generation.data[0].output["tool_calls"][0]["function"]["arguments"] is not None - ) - assert generation.data[0].input["tools"] == tools - assert generation.data[0].input["messages"] == messages + assert generation[0].output["tool_calls"][0]["function"]["arguments"] is not None + assert generation[0].input["tools"] == tools + assert generation[0].input["messages"] == messages def test_langchain_integration(openai): @@ -1047,28 +999,28 @@ def test_structured_output_response_format_kwarg(openai): langfuse.flush() - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" + generation = wait_for_observations( + name=generation_name, type="GENERATION", expand_metadata="response_format" ) - assert len(generation.data) != 0 - assert generation.data[0].name == generation_name - assert generation.data[0].metadata["someKey"] == "someResponse" - assert generation.data[0].metadata["response_format"] == { + assert len(generation) != 0 + assert generation[0].name == generation_name + assert generation[0].metadata["someKey"] == "someResponse" + assert generation[0].metadata["response_format"] == { "type": "json_schema", "json_schema": json_schema, } - assert generation.data[0].input == [ + assert generation[0].input == [ {"role": "system", "content": "You are a helpful math tutor."}, {"content": "solve 8x + 31 = 2", "role": "user"}, ] - assert generation.data[0].type == "GENERATION" - assert generation.data[0].model == "gpt-4o-2024-08-06" - assert generation.data[0].start_time is not None - assert generation.data[0].end_time is not None - assert generation.data[0].start_time < generation.data[0].end_time - assert generation.data[0].model_parameters == { + assert generation[0].type == "GENERATION" + assert generation[0].model == "gpt-4o-2024-08-06" + assert generation[0].start_time is not None + assert generation[0].end_time is not None + assert generation[0].start_time < generation[0].end_time + assert generation[0].model_parameters == { "service_tier": "default", "temperature": 1, "top_p": 1, @@ -1076,10 +1028,10 @@ def test_structured_output_response_format_kwarg(openai): "max_tokens": "Infinity", "presence_penalty": 0, } - assert generation.data[0].usage.input is not None - assert generation.data[0].usage.output is not None - assert generation.data[0].usage.total is not None - assert generation.data[0].output["role"] == "assistant" + assert generation[0].usage_details["input"] is not None + assert generation[0].usage_details["output"] is not None + assert generation[0].usage_details["total"] is not None + assert generation[0].output["role"] == "assistant" def test_structured_output_beta_completions_parse(openai): @@ -1117,31 +1069,29 @@ class CalendarEvent(BaseModel): if Version(openai.__version__) >= Version("1.50.0"): # Check the trace and observation properties - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) + generation = wait_for_observations(name=generation_name, type="GENERATION") - assert len(generation.data) == 1 - assert generation.data[0].name == generation_name - assert generation.data[0].type == "GENERATION" - assert "gpt-4o" in generation.data[0].model - assert generation.data[0].start_time is not None - assert generation.data[0].end_time is not None - assert generation.data[0].start_time < generation.data[0].end_time + assert len(generation) == 1 + assert generation[0].name == generation_name + assert generation[0].type == "GENERATION" + assert "gpt-4o" in generation[0].model + assert generation[0].start_time is not None + assert generation[0].end_time is not None + assert generation[0].start_time < generation[0].end_time # Check input and output - assert len(generation.data[0].input) == 2 - assert generation.data[0].input[0]["role"] == "system" - assert generation.data[0].input[1]["role"] == "user" - assert isinstance(generation.data[0].output, dict) - assert "name" in generation.data[0].output["content"] - assert "date" in generation.data[0].output["content"] - assert "participants" in generation.data[0].output["content"] + assert len(generation[0].input) == 2 + assert generation[0].input[0]["role"] == "system" + assert generation[0].input[1]["role"] == "user" + assert isinstance(generation[0].output, dict) + assert "name" in generation[0].output["content"] + assert "date" in generation[0].output["content"] + assert "participants" in generation[0].output["content"] # Check usage - assert generation.data[0].usage.input is not None - assert generation.data[0].usage.output is not None - assert generation.data[0].usage.total is not None + assert generation[0].usage_details["input"] is not None + assert generation[0].usage_details["output"] is not None + assert generation[0].usage_details["total"] is not None @pytest.mark.asyncio @@ -1163,19 +1113,17 @@ async def test_close_async_stream(openai): langfuse.flush() - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) - - assert len(generation.data) != 0 - assert generation.data[0].name == generation_name - assert generation.data[0].input == [{"content": "1 + 1 = ", "role": "user"}] - assert generation.data[0].type == "GENERATION" - assert generation.data[0].model == "gpt-3.5-turbo-0125" - assert generation.data[0].start_time is not None - assert generation.data[0].end_time is not None - assert generation.data[0].start_time < generation.data[0].end_time - assert generation.data[0].model_parameters == { + generation = wait_for_observations(name=generation_name, type="GENERATION") + + assert len(generation) != 0 + assert generation[0].name == generation_name + assert generation[0].input == [{"content": "1 + 1 = ", "role": "user"}] + assert generation[0].type == "GENERATION" + assert generation[0].model == "gpt-3.5-turbo-0125" + assert generation[0].start_time is not None + assert generation[0].end_time is not None + assert generation[0].start_time < generation[0].end_time + assert generation[0].model_parameters == { "service_tier": "default", "temperature": 1, "top_p": 1, @@ -1183,15 +1131,15 @@ async def test_close_async_stream(openai): "max_tokens": "Infinity", "presence_penalty": 0, } - assert generation.data[0].usage.input is not None - assert generation.data[0].usage.output is not None - assert generation.data[0].usage.total is not None - assert "2" in str(generation.data[0].output) + assert generation[0].usage_details["input"] is not None + assert generation[0].usage_details["output"] is not None + assert generation[0].usage_details["total"] is not None + assert "2" in str(generation[0].output) # Completion start time for time-to-first-token - assert generation.data[0].completion_start_time is not None - assert generation.data[0].completion_start_time >= generation.data[0].start_time - assert generation.data[0].completion_start_time <= generation.data[0].end_time + assert generation[0].completion_start_time is not None + assert generation[0].completion_start_time >= generation[0].start_time + assert generation[0].completion_start_time <= generation[0].end_time def test_base_64_image_input(openai): @@ -1225,26 +1173,24 @@ def test_base_64_image_input(openai): langfuse.flush() - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) + generation = wait_for_observations(name=generation_name, type="GENERATION") - assert len(generation.data) != 0 - assert generation.data[0].name == generation_name - assert generation.data[0].input[0]["content"][0]["text"] == "What’s in this image?" + assert len(generation) != 0 + assert generation[0].name == generation_name + assert generation[0].input[0]["content"][0]["text"] == "What’s in this image?" assert ( f"@@@langfuseMedia:type={content_type}|id=" - in generation.data[0].input[0]["content"][1]["image_url"]["url"] + in generation[0].input[0]["content"][1]["image_url"]["url"] ) - assert generation.data[0].type == "GENERATION" - assert "gpt-4o-mini" in generation.data[0].model - assert generation.data[0].start_time is not None - assert generation.data[0].end_time is not None - assert generation.data[0].start_time < generation.data[0].end_time - assert generation.data[0].usage.input is not None - assert generation.data[0].usage.output is not None - assert generation.data[0].usage.total is not None - assert "dog" in generation.data[0].output["content"] + assert generation[0].type == "GENERATION" + assert "gpt-4o-mini" in generation[0].model + assert generation[0].start_time is not None + assert generation[0].end_time is not None + assert generation[0].start_time < generation[0].end_time + assert generation[0].usage_details["input"] is not None + assert generation[0].usage_details["output"] is not None + assert generation[0].usage_details["total"] is not None + assert "dog" in generation[0].output["content"] def test_audio_input_and_output(openai): @@ -1277,32 +1223,28 @@ def test_audio_input_and_output(openai): langfuse.flush() - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) + generation = wait_for_observations(name=generation_name, type="GENERATION") - assert len(generation.data) != 0 - assert generation.data[0].name == generation_name + assert len(generation) != 0 + assert generation[0].name == generation_name assert ( - generation.data[0].input[0]["content"][0]["text"] - == "Do what this recording says." + generation[0].input[0]["content"][0]["text"] == "Do what this recording says." ) assert ( "@@@langfuseMedia:type=audio/wav|id=" - in generation.data[0].input[0]["content"][1]["input_audio"]["data"] - ) - assert generation.data[0].type == "GENERATION" - assert generation.data[0].model == model - assert generation.data[0].start_time is not None - assert generation.data[0].end_time is not None - assert generation.data[0].start_time < generation.data[0].end_time - assert generation.data[0].usage.input is not None - assert generation.data[0].usage.output is not None - assert generation.data[0].usage.total is not None - print(generation.data[0].output) + in generation[0].input[0]["content"][1]["input_audio"]["data"] + ) + assert generation[0].type == "GENERATION" + assert generation[0].model == model + assert generation[0].start_time is not None + assert generation[0].end_time is not None + assert generation[0].start_time < generation[0].end_time + assert generation[0].usage_details["input"] is not None + assert generation[0].usage_details["output"] is not None + assert generation[0].usage_details["total"] is not None + print(generation[0].output) assert ( - "@@@langfuseMedia:type=audio/wav|id=" - in generation.data[0].output["audio"]["data"] + "@@@langfuseMedia:type=audio/wav|id=" in generation[0].output["audio"]["data"] ) @@ -1317,25 +1259,22 @@ def test_response_api_text_input(openai): ) langfuse.flush() - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) + generation = wait_for_observations(name=generation_name, type="GENERATION") - assert len(generation.data) != 0 - generationData = generation.data[0] + assert len(generation) != 0 + generationData = generation[0] assert generationData.name == generation_name assert ( - generation.data[0].input - == "Tell me a three sentence bedtime story about a unicorn." + generation[0].input == "Tell me a three sentence bedtime story about a unicorn." ) assert generationData.type == "GENERATION" assert "gpt-4o" in generationData.model assert generationData.start_time is not None assert generationData.end_time is not None assert generationData.start_time < generationData.end_time - assert generationData.usage.input is not None - assert generationData.usage.output is not None - assert generationData.usage.total is not None + assert generationData.usage_details["input"] is not None + assert generationData.usage_details["output"] is not None + assert generationData.usage_details["total"] is not None assert generationData.output is not None @@ -1363,22 +1302,20 @@ def test_response_api_image_input(openai): langfuse.flush() - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) + generation = wait_for_observations(name=generation_name, type="GENERATION") - assert len(generation.data) != 0 - generationData = generation.data[0] + assert len(generation) != 0 + generationData = generation[0] assert generationData.name == generation_name - assert generation.data[0].input[0]["content"][0]["text"] == "what is in this image?" + assert generation[0].input[0]["content"][0]["text"] == "what is in this image?" assert generationData.type == "GENERATION" assert "gpt-4o" in generationData.model assert generationData.start_time is not None assert generationData.end_time is not None assert generationData.start_time < generationData.end_time - assert generationData.usage.input is not None - assert generationData.usage.output is not None - assert generationData.usage.total is not None + assert generationData.usage_details["input"] is not None + assert generationData.usage_details["output"] is not None + assert generationData.usage_details["total"] is not None assert generationData.output is not None @@ -1395,12 +1332,10 @@ def test_response_api_web_search(openai): langfuse.flush() - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) + generation = wait_for_observations(name=generation_name, type="GENERATION") - assert len(generation.data) != 0 - generationData = generation.data[0] + assert len(generation) != 0 + generationData = generation[0] assert generationData.name == generation_name assert generationData.input == { "input": "What was a positive news story from today?", @@ -1411,9 +1346,9 @@ def test_response_api_web_search(openai): assert generationData.start_time is not None assert generationData.end_time is not None assert generationData.start_time < generationData.end_time - assert generationData.usage.input is not None - assert generationData.usage.output is not None - assert generationData.usage.total is not None + assert generationData.usage_details["input"] is not None + assert generationData.usage_details["output"] is not None + assert generationData.usage_details["total"] is not None assert generationData.output is not None assert generationData.metadata is not None @@ -1435,14 +1370,12 @@ def test_response_api_streaming(openai): langfuse.flush() - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) + generation = wait_for_observations(name=generation_name, type="GENERATION") - assert len(generation.data) != 0 - generationData = generation.data[0] + assert len(generation) != 0 + generationData = generation[0] assert generationData.name == generation_name - assert generation.data[0].input == [ + assert generation[0].input == [ {"role": "system", "content": "You are a helpful assistant."}, {"role": "user", "content": "Hello!"}, ] @@ -1451,9 +1384,9 @@ def test_response_api_streaming(openai): assert generationData.start_time is not None assert generationData.end_time is not None assert generationData.start_time < generationData.end_time - assert generationData.usage.input is not None - assert generationData.usage.output is not None - assert generationData.usage.total is not None + assert generationData.usage_details["input"] is not None + assert generationData.usage_details["output"] is not None + assert generationData.usage_details["total"] is not None assert generationData.output is not None assert generationData.metadata is not None assert generationData.metadata["instructions"] == "You are a helpful assistant." @@ -1492,14 +1425,12 @@ def test_response_api_functions(openai): langfuse.flush() - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) + generation = wait_for_observations(name=generation_name, type="GENERATION") - assert len(generation.data) != 0 - generationData = generation.data[0] + assert len(generation) != 0 + generationData = generation[0] assert generationData.name == generation_name - assert generation.data[0].input == { + assert generation[0].input == { "input": "What is the weather like in Boston today?", "tools": tools, "tool_choice": "auto", @@ -1509,9 +1440,9 @@ def test_response_api_functions(openai): assert generationData.start_time is not None assert generationData.end_time is not None assert generationData.start_time < generationData.end_time - assert generationData.usage.input is not None - assert generationData.usage.output is not None - assert generationData.usage.total is not None + assert generationData.usage_details["input"] is not None + assert generationData.usage_details["output"] is not None + assert generationData.usage_details["total"] is not None assert generationData.output is not None assert generationData.metadata is not None @@ -1528,22 +1459,20 @@ def test_response_api_reasoning(openai): ) langfuse.flush() - generation = get_api().legacy.observations_v1.get_many( - name=generation_name, type="GENERATION" - ) + generation = wait_for_observations(name=generation_name, type="GENERATION") - assert len(generation.data) != 0 - generationData = generation.data[0] + assert len(generation) != 0 + generationData = generation[0] assert generationData.name == generation_name - assert generation.data[0].input == "How much wood would a woodchuck chuck?" + assert generation[0].input == "How much wood would a woodchuck chuck?" assert generationData.type == "GENERATION" assert "o3-mini" in generationData.model assert generationData.start_time is not None assert generationData.end_time is not None assert generationData.start_time < generationData.end_time - assert generationData.usage.input is not None - assert generationData.usage.output is not None - assert generationData.usage.total is not None + assert generationData.usage_details["input"] is not None + assert generationData.usage_details["output"] is not None + assert generationData.usage_details["total"] is not None assert generationData.output is not None assert generationData.metadata is not None @@ -1560,12 +1489,10 @@ def test_openai_embeddings(openai): langfuse.flush() sleep(1) - embedding = get_api().legacy.observations_v1.get_many( - name=embedding_name, type="EMBEDDING" - ) + embedding = wait_for_observations(name=embedding_name, type="EMBEDDING") - assert len(embedding.data) != 0 - embedding_data = embedding.data[0] + assert len(embedding) != 0 + embedding_data = embedding[0] assert embedding_data.name == embedding_name assert embedding_data.metadata["test_key"] == "test_value" assert embedding_data.input == "The quick brown fox jumps over the lazy dog" @@ -1574,8 +1501,8 @@ def test_openai_embeddings(openai): assert embedding_data.start_time is not None assert embedding_data.end_time is not None assert embedding_data.start_time < embedding_data.end_time - assert embedding_data.usage.input is not None - assert embedding_data.usage.total is not None + assert embedding_data.usage_details["input"] is not None + assert embedding_data.usage_details["total"] is not None assert embedding_data.output is not None assert "dimensions" in embedding_data.output assert "count" in embedding_data.output @@ -1596,18 +1523,16 @@ def test_openai_embeddings_multiple_inputs(openai): langfuse.flush() sleep(1) - embedding = get_api().legacy.observations_v1.get_many( - name=embedding_name, type="EMBEDDING" - ) + embedding = wait_for_observations(name=embedding_name, type="EMBEDDING") - assert len(embedding.data) != 0 - embedding_data = embedding.data[0] + assert len(embedding) != 0 + embedding_data = embedding[0] assert embedding_data.name == embedding_name assert embedding_data.input == inputs assert embedding_data.type == "EMBEDDING" assert "text-embedding-ada-002" in embedding_data.model - assert embedding_data.usage.input is not None - assert embedding_data.usage.total is not None + assert embedding_data.usage_details["input"] is not None + assert embedding_data.usage_details["total"] is not None assert embedding_data.output["count"] == len(inputs) @@ -1629,16 +1554,14 @@ async def test_async_openai_embeddings(openai): langfuse.flush() sleep(1) - embedding = get_api().legacy.observations_v1.get_many( - name=embedding_name, type="EMBEDDING" - ) + embedding = wait_for_observations(name=embedding_name, type="EMBEDDING") - assert len(embedding.data) != 0 - embedding_data = embedding.data[0] + assert len(embedding) != 0 + embedding_data = embedding[0] assert embedding_data.name == embedding_name assert embedding_data.input == "Async embedding test" assert embedding_data.type == "EMBEDDING" assert "text-embedding-ada-002" in embedding_data.model assert embedding_data.metadata["async"] is True - assert embedding_data.usage.input is not None - assert embedding_data.usage.total is not None + assert embedding_data.usage_details["input"] is not None + assert embedding_data.usage_details["total"] is not None diff --git a/tests/support/api_wrapper.py b/tests/support/api_wrapper.py deleted file mode 100644 index c4519252f..000000000 --- a/tests/support/api_wrapper.py +++ /dev/null @@ -1,130 +0,0 @@ -import os - -import httpx - -from langfuse.api.commons.errors.not_found_error import NotFoundError -from tests.support.retry import ( - DEFAULT_RETRY_INTERVAL_SECONDS, - DEFAULT_RETRY_TIMEOUT_SECONDS, - is_not_found_payload, - retry_until_ready, -) - - -class LangfuseAPI: - def __init__(self, username=None, password=None, base_url=None): - username = username if username else os.environ["LANGFUSE_PUBLIC_KEY"] - password = password if password else os.environ["LANGFUSE_SECRET_KEY"] - self.auth = (username, password) - self.BASE_URL = base_url if base_url else os.environ["LANGFUSE_BASE_URL"] - - def _get_json( - self, - url, - params=None, - *, - retry=True, - is_result_ready=None, - timeout_seconds=DEFAULT_RETRY_TIMEOUT_SECONDS, - interval_seconds=DEFAULT_RETRY_INTERVAL_SECONDS, - ): - def _request(): - response = httpx.get(url, params=params, auth=self.auth) - payload = response.json() - - if response.status_code == 404 and is_not_found_payload(payload): - raise NotFoundError(body=payload, headers=dict(response.headers)) - - return payload - - if not retry: - return _request() - - return retry_until_ready( - _request, - is_result_ready=is_result_ready, - timeout_seconds=timeout_seconds, - interval_seconds=interval_seconds, - ) - - def get_observation( - self, - observation_id, - *, - retry=True, - is_result_ready=None, - timeout_seconds=DEFAULT_RETRY_TIMEOUT_SECONDS, - interval_seconds=DEFAULT_RETRY_INTERVAL_SECONDS, - ): - url = f"{self.BASE_URL}/api/public/observations/{observation_id}" - return self._get_json( - url, - retry=retry, - is_result_ready=is_result_ready, - timeout_seconds=timeout_seconds, - interval_seconds=interval_seconds, - ) - - def get_scores( - self, - page=None, - limit=None, - user_id=None, - name=None, - *, - retry=True, - is_result_ready=None, - timeout_seconds=DEFAULT_RETRY_TIMEOUT_SECONDS, - interval_seconds=DEFAULT_RETRY_INTERVAL_SECONDS, - ): - params = {"page": page, "limit": limit, "userId": user_id, "name": name} - url = f"{self.BASE_URL}/api/public/scores" - return self._get_json( - url, - params=params, - retry=retry, - is_result_ready=is_result_ready, - timeout_seconds=timeout_seconds, - interval_seconds=interval_seconds, - ) - - def get_traces( - self, - page=None, - limit=None, - user_id=None, - name=None, - *, - retry=True, - is_result_ready=None, - timeout_seconds=DEFAULT_RETRY_TIMEOUT_SECONDS, - interval_seconds=DEFAULT_RETRY_INTERVAL_SECONDS, - ): - params = {"page": page, "limit": limit, "userId": user_id, "name": name} - url = f"{self.BASE_URL}/api/public/traces" - return self._get_json( - url, - params=params, - retry=retry, - is_result_ready=is_result_ready, - timeout_seconds=timeout_seconds, - interval_seconds=interval_seconds, - ) - - def get_trace( - self, - trace_id, - *, - retry=True, - is_result_ready=None, - timeout_seconds=DEFAULT_RETRY_TIMEOUT_SECONDS, - interval_seconds=DEFAULT_RETRY_INTERVAL_SECONDS, - ): - url = f"{self.BASE_URL}/api/public/traces/{trace_id}" - return self._get_json( - url, - retry=retry, - is_result_ready=is_result_ready, - timeout_seconds=timeout_seconds, - interval_seconds=interval_seconds, - ) diff --git a/tests/support/utils.py b/tests/support/utils.py index a29274d3a..191eeae05 100644 --- a/tests/support/utils.py +++ b/tests/support/utils.py @@ -1,19 +1,42 @@ import base64 +import json import os -from typing import Any, Callable, TypeVar +from dataclasses import dataclass +from typing import Any, Callable, Sequence, TypeVar from uuid import uuid4 -from langfuse.api import LangfuseAPI +from langfuse.api import LangfuseAPI, ObservationV2, ScoreV3 from tests.support.retry import ( DEFAULT_RETRY_INTERVAL_SECONDS, DEFAULT_RETRY_TIMEOUT_SECONDS, retry_until_ready, ) -READ_METHOD_NAMES = {"get", "get_by_id", "get_many", "get_run", "list"} -PAGINATION_ARGUMENTS = {"limit", "page"} +READ_METHOD_NAMES = {"get", "get_by_id", "get_many", "get_many_v3", "get_run", "list"} +PAGINATION_ARGUMENTS = {"limit", "page", "cursor", "fields", "expand_metadata"} T = TypeVar("T") +ALL_OBSERVATION_FIELDS = ( + "core,basic,time,io,metadata,model,usage,prompt,metrics,trace_context" +) +SCORE_FIELDS = "details,subject" + +# The v2 observations API returns "" instead of null for unset string fields. +_EMPTY_AS_NONE_FIELDS = ( + "name", + "status_message", + "version", + "user_id", + "session_id", + "model", + "internal_model_id", + "prompt_id", + "prompt_name", + "trace_name", + "release", +) +_SDK_METADATA_KEY_PREFIXES = ("scope.", "resourceAttributes.") + def _has_filters(kwargs: dict[str, Any]) -> bool: return any( @@ -51,7 +74,9 @@ def _call(*args: Any, **kwargs: Any) -> Any: def _result_ready(method_name: str, kwargs: dict[str, Any]): - if method_name not in {"get_many", "list"} or not _has_filters(kwargs): + if method_name not in {"get_many", "get_many_v3", "list"} or not _has_filters( + kwargs + ): return None def _has_data(result: Any) -> bool: @@ -89,17 +114,315 @@ def wait_for_result( ) -def wait_for_trace( +def _parse_json_string(value: Any) -> Any: + # Only objects and arrays: the SDK sends string IO unquoted, so text such as + # "2" or "true" must stay a string. + if not isinstance(value, str) or not value.lstrip().startswith(("{", "[")): + return value + + try: + return json.loads(value) + except ValueError: + return value + + +def normalize_observation(observation: ObservationV2) -> ObservationV2: + """Map a v2 observation to the shape the SDK sent. + + The v2 API returns input/output as raw strings and unset string fields as + "". This parses JSON object/array input/output and maps "" to None. + """ + update: dict[str, Any] = { + "input": _parse_json_string(observation.input), + "output": _parse_json_string(observation.output), + } + for field in _EMPTY_AS_NONE_FIELDS: + if getattr(observation, field, None) == "": + update[field] = None + + return observation.model_copy(update=update) + + +def user_metadata(observation: ObservationV2) -> dict[str, Any]: + """Observation metadata without the scope/resource keys the server adds.""" + metadata = observation.metadata or {} + assert isinstance(metadata, dict), metadata + + return { + key: value + for key, value in metadata.items() + if not key.startswith(_SDK_METADATA_KEY_PREFIXES) + } + + +def get_observations( + *, + fields: str = ALL_OBSERVATION_FIELDS, + api: Any = None, + **filters: Any, +) -> list[ObservationV2]: + """Fetch all observations matching `filters` (all pages), oldest first.""" + api = api or get_api(retry=False) + observations: list[ObservationV2] = [] + cursor = None + + while True: + response = api.observations.get_many( + fields=fields, limit=1000, cursor=cursor, **filters + ) + observations.extend(response.data) + cursor = response.meta.cursor + if not cursor or not response.data: + break + + # The events table can briefly return several rows for one span until + # ClickHouse merges them; keep the most recently updated row per id. + latest_by_id: dict[str, ObservationV2] = {} + for observation in observations: + current = latest_by_id.get(observation.id) + if current is None or (observation.updated_at or observation.start_time) >= ( + current.updated_at or current.start_time + ): + latest_by_id[observation.id] = observation + + return sorted( + (normalize_observation(observation) for observation in latest_by_id.values()), + key=lambda observation: observation.start_time, + ) + + +def wait_for_observations( + trace_id: str | None = None, + *, + min_count: int = 1, + is_result_ready: Callable[[list[ObservationV2]], bool] | None = None, + fields: str = ALL_OBSERVATION_FIELDS, + timeout_seconds: float = DEFAULT_RETRY_TIMEOUT_SECONDS, + interval_seconds: float = DEFAULT_RETRY_INTERVAL_SECONDS, + **filters: Any, +) -> list[ObservationV2]: + """Poll the v2 observations API until at least `min_count` observations + match (and `is_result_ready` holds), then return them oldest first.""" + if trace_id is not None: + filters["trace_id"] = trace_id + assert filters, "wait_for_observations needs at least one filter" + + def _ready(observations: list[ObservationV2]) -> bool: + return len(observations) >= min_count and ( + is_result_ready is None or is_result_ready(observations) + ) + + return wait_for_result( + lambda: get_observations(fields=fields, **filters), + is_result_ready=_ready, + timeout_seconds=timeout_seconds, + interval_seconds=interval_seconds, + ) + + +def get_root_observation(observations: Sequence[ObservationV2]) -> ObservationV2: + """Return the single root observation, which carries the trace-level + attributes (trace name, user, session, tags, public) and trace IO.""" + roots = [ + observation for observation in observations if observation.is_root_observation + ] + assert len(roots) == 1, ( + f"expected exactly one root observation, got {[r.name for r in roots]}" + ) + + return roots[0] + + +def wait_for_root_observation( + trace_id: str, + *, + min_count: int = 1, + is_result_ready: Callable[[ObservationV2], bool] | None = None, + timeout_seconds: float = DEFAULT_RETRY_TIMEOUT_SECONDS, + interval_seconds: float = DEFAULT_RETRY_INTERVAL_SECONDS, +) -> ObservationV2: + def _ready(observations: list[ObservationV2]) -> bool: + roots = [o for o in observations if o.is_root_observation] + return len(roots) == 1 and ( + is_result_ready is None or is_result_ready(roots[0]) + ) + + return get_root_observation( + wait_for_observations( + trace_id, + min_count=min_count, + is_result_ready=_ready, + timeout_seconds=timeout_seconds, + interval_seconds=interval_seconds, + ) + ) + + +@dataclass(frozen=True) +class TraceSnapshot: + """A trace as v4 exposes it: its observations plus (optionally) scores. + + Mirrors how the platform aggregates events into a trace: input, output + and metadata come from the root observation; name, user, session, + version, release and environment are the latest non-empty value across + all observations; tags are the union; public is true if any observation + is public. + """ + + id: str + observations: list[ObservationV2] + scores: list[ScoreV3] | None = None + + @property + def root(self) -> ObservationV2: + return get_root_observation(self.observations) + + def _latest(self, field: str) -> Any: + values = [ + getattr(observation, field) + for observation in self.observations + if getattr(observation, field) + ] + return values[-1] if values else None + + @property + def name(self) -> str | None: + # The API falls back to the root's own name when no trace name was + # set, so a root whose trace_name equals its name carries no signal. + explicit_names = [ + observation.trace_name + for observation in self.observations + if observation.trace_name + and not ( + observation.is_root_observation + and observation.trace_name == observation.name + ) + ] + return explicit_names[-1] if explicit_names else self.root.trace_name + + @property + def user_id(self) -> str | None: + return self._latest("user_id") + + @property + def session_id(self) -> str | None: + return self._latest("session_id") + + @property + def tags(self) -> list[str]: + return sorted( + {tag for observation in self.observations for tag in observation.tags or []} + ) + + @property + def public(self) -> bool: + return any(observation.public for observation in self.observations) + + @property + def version(self) -> str | None: + return self._latest("version") + + @property + def release(self) -> str | None: + return self._latest("release") + + @property + def environment(self) -> str | None: + return self._latest("environment") + + @property + def input(self) -> Any: + return self.root.input + + @property + def output(self) -> Any: + return self.root.output + + @property + def metadata(self) -> dict[str, Any]: + return user_metadata(self.root) + + +def wait_for_trace_snapshot( trace_id: str, *, - is_result_ready: Callable[[Any], bool] | None = None, + min_observations: int = 1, + min_scores: int | None = None, + is_result_ready: Callable[[TraceSnapshot], bool] | None = None, timeout_seconds: float = DEFAULT_RETRY_TIMEOUT_SECONDS, interval_seconds: float = DEFAULT_RETRY_INTERVAL_SECONDS, -): - api = get_api(retry=False) +) -> TraceSnapshot: + """Poll until the trace has `min_observations` observations (and + `min_scores` scores, which are only fetched when given).""" + + def _fetch() -> TraceSnapshot: + return TraceSnapshot( + id=trace_id, + observations=get_observations(trace_id=trace_id), + scores=None if min_scores is None else get_scores(trace_id=trace_id), + ) + + def _ready(snapshot: TraceSnapshot) -> bool: + if len(snapshot.observations) < min_observations: + return False + if min_scores is not None and len(snapshot.scores or []) < min_scores: + return False + if is_result_ready is None: + return True + try: + return is_result_ready(snapshot) + except AssertionError: + # Root-derived attributes are unavailable until the root arrives. + return False + return wait_for_result( - lambda: api.trace.get(trace_id), - is_result_ready=is_result_ready, + _fetch, + is_result_ready=_ready, + timeout_seconds=timeout_seconds, + interval_seconds=interval_seconds, + ) + + +def get_scores( + *, fields: str = SCORE_FIELDS, api: Any = None, **filters: Any +) -> list[ScoreV3]: + api = api or get_api(retry=False) + scores: list[ScoreV3] = [] + cursor = None + + while True: + response = api.scores_v3.get_many_v3( + fields=fields, limit=100, cursor=cursor, **filters + ) + scores.extend(response.data) + cursor = response.meta.cursor + if not cursor or not response.data: + break + + return scores + + +def wait_for_scores( + *, + min_count: int = 1, + is_result_ready: Callable[[list[ScoreV3]], bool] | None = None, + fields: str = SCORE_FIELDS, + timeout_seconds: float = DEFAULT_RETRY_TIMEOUT_SECONDS, + interval_seconds: float = DEFAULT_RETRY_INTERVAL_SECONDS, + **filters: Any, +) -> list[ScoreV3]: + """Poll the v3 scores API (filters: trace_id, session_id, observation_id, + name, ...) until at least `min_count` scores match.""" + assert filters, "wait_for_scores needs at least one filter" + + def _ready(scores: list[ScoreV3]) -> bool: + return len(scores) >= min_count and ( + is_result_ready is None or is_result_ready(scores) + ) + + return wait_for_result( + lambda: get_scores(fields=fields, **filters), + is_result_ready=_ready, timeout_seconds=timeout_seconds, interval_seconds=interval_seconds, ) diff --git a/tests/unit/test_e2e_support.py b/tests/unit/test_e2e_support.py index 8320bd2fe..98d25127c 100644 --- a/tests/unit/test_e2e_support.py +++ b/tests/unit/test_e2e_support.py @@ -1,144 +1,260 @@ +from datetime import datetime, timedelta, timezone from types import SimpleNamespace +from langfuse.api import ObservationV2 from langfuse.api.commons.errors.not_found_error import NotFoundError -from tests.support.api_wrapper import LangfuseAPI as SupportLangfuseAPI from tests.support.retry import retry_until_ready -from tests.support.utils import get_api, wait_for_trace +from tests.support.utils import ( + TraceSnapshot, + get_api, + get_observations, + normalize_observation, + user_metadata, + wait_for_observations, + wait_for_scores, + wait_for_trace_snapshot, +) +START = datetime(2024, 1, 1, tzinfo=timezone.utc) -def test_get_api_retries_not_found(monkeypatch): - monkeypatch.setattr("tests.support.retry.sleep", lambda _: None) - attempts = {"count": 0} +def _observation(index: int = 0, **fields) -> ObservationV2: + defaults = { + "id": f"obs-{index}", + "trace_id": "trace-123", + "start_time": START + timedelta(seconds=index), + "project_id": "project", + "type": "SPAN", + "is_root_observation": False, + } + return ObservationV2(**{**defaults, **fields}) - class FakeTraceService: - def get(self, trace_id): - attempts["count"] += 1 - if attempts["count"] < 3: - raise NotFoundError( - body={ - "error": "LangfuseNotFoundError", - "message": f"Trace {trace_id} not found within authorized project", - } - ) +def _page(data, cursor=None): + return SimpleNamespace(data=data, meta=SimpleNamespace(cursor=cursor)) - return {"id": trace_id} - class FakeClient: - trace = FakeTraceService() +def _install_client(monkeypatch, **services): + monkeypatch.setattr("tests.support.retry.sleep", lambda _: None) + client = SimpleNamespace(**services) + monkeypatch.setattr("tests.support.utils.LangfuseAPI", lambda **_: client) - monkeypatch.setattr("tests.support.utils.LangfuseAPI", lambda **_: FakeClient()) - trace = get_api().trace.get("trace-123") +def test_get_api_retries_not_found(monkeypatch): + attempts = {"count": 0} - assert trace == {"id": "trace-123"} - assert attempts["count"] == 3 + def get_many(**kwargs): + attempts["count"] += 1 + if attempts["count"] < 3: + raise NotFoundError( + body={ + "error": "LangfuseNotFoundError", + "message": "Observations not found within authorized project", + } + ) -def test_get_api_retries_filtered_lists(monkeypatch): - monkeypatch.setattr("tests.support.retry.sleep", lambda _: None) + return _page([kwargs["trace_id"]]) - attempts = {"count": 0} + _install_client(monkeypatch, observations=SimpleNamespace(get_many=get_many)) - class FakeTraceService: - def list(self, **kwargs): - attempts["count"] += 1 + response = get_api().observations.get_many(trace_id="trace-123") - if attempts["count"] < 3: - return SimpleNamespace(data=[]) + assert response.data == ["trace-123"] + assert attempts["count"] == 3 - return SimpleNamespace(data=[kwargs["name"]]) - class FakeClient: - trace = FakeTraceService() +def test_get_api_retries_filtered_lists(monkeypatch): + attempts = {"count": 0} - monkeypatch.setattr("tests.support.utils.LangfuseAPI", lambda **_: FakeClient()) + def get_many(**kwargs): + attempts["count"] += 1 + return _page([] if attempts["count"] < 3 else [kwargs["name"]]) - response = get_api().trace.list(name="ready-trace") + _install_client(monkeypatch, observations=SimpleNamespace(get_many=get_many)) - assert response.data == ["ready-trace"] + response = get_api().observations.get_many(name="ready-observation") + + assert response.data == ["ready-observation"] assert attempts["count"] == 3 def test_get_api_retry_can_be_disabled(monkeypatch): attempts = {"count": 0} - class FakeTraceService: - def list(self, **kwargs): - attempts["count"] += 1 - return SimpleNamespace(data=[]) - - class FakeClient: - trace = FakeTraceService() + def get_many(**kwargs): + attempts["count"] += 1 + return _page([]) - monkeypatch.setattr("tests.support.utils.LangfuseAPI", lambda **_: FakeClient()) + _install_client(monkeypatch, observations=SimpleNamespace(get_many=get_many)) - response = get_api(retry=False).trace.list(name="missing-trace") + response = get_api(retry=False).observations.get_many(name="missing") assert response.data == [] assert attempts["count"] == 1 -def test_raw_api_wrapper_retries_not_found_payload(monkeypatch): - monkeypatch.setattr("tests.support.retry.sleep", lambda _: None) +def test_normalize_observation_parses_io_and_maps_empty_strings_to_none(): + observation = normalize_observation( + _observation( + input='{"question": "hi"}', + output="plain text", + name="", + session_id="", + user_id="user-1", + ) + ) - attempts = {"count": 0} + assert observation.input == {"question": "hi"} + assert observation.output == "plain text" + assert observation.name is None + assert observation.session_id is None + assert observation.user_id == "user-1" - class FakeResponse: - def __init__(self, status_code, payload): - self.status_code = status_code - self._payload = payload - self.headers = {} - def json(self): - return self._payload +def test_user_metadata_drops_server_added_keys(): + observation = _observation( + metadata={ + "key": "value", + "scope.name": "langfuse-sdk", + "resourceAttributes.service.name": "test", + } + ) - def fake_get(*args, **kwargs): - attempts["count"] += 1 + assert user_metadata(observation) == {"key": "value"} - if attempts["count"] < 3: - return FakeResponse( - 404, - { - "error": "LangfuseNotFoundError", - "message": "Trace trace-123 not found within authorized project", - }, - ) - return FakeResponse(200, {"id": "trace-123", "observations": []}) +def test_get_observations_follows_cursor_and_sorts_by_start_time(monkeypatch): + calls = [] + pages = { + None: _page([_observation(2), _observation(0)], cursor="next"), + "next": _page([_observation(1)]), + } - monkeypatch.setattr("tests.support.api_wrapper.httpx.get", fake_get) + def get_many(**kwargs): + calls.append(kwargs) + return pages[kwargs["cursor"]] - api = SupportLangfuseAPI(username="user", password="pass", base_url="http://test") - trace = api.get_trace("trace-123") + _install_client(monkeypatch, observations=SimpleNamespace(get_many=get_many)) - assert trace["id"] == "trace-123" - assert attempts["count"] == 3 + observations = get_observations(trace_id="trace-123") + assert [o.id for o in observations] == ["obs-0", "obs-1", "obs-2"] + assert [call["cursor"] for call in calls] == [None, "next"] + assert all(call["trace_id"] == "trace-123" for call in calls) -def test_wait_for_trace_retries_until_predicate_matches(monkeypatch): - monkeypatch.setattr("tests.support.retry.sleep", lambda _: None) +def test_wait_for_observations_polls_until_min_count(monkeypatch): attempts = {"count": 0} - class FakeTraceService: - def get(self, trace_id): - attempts["count"] += 1 - return {"id": trace_id, "observations": [1] * attempts["count"]} + def get_many(**kwargs): + attempts["count"] += 1 + return _page([_observation(i) for i in range(attempts["count"])]) + + _install_client(monkeypatch, observations=SimpleNamespace(get_many=get_many)) - class FakeClient: - trace = FakeTraceService() + observations = wait_for_observations("trace-123", min_count=3) + + assert len(observations) == 3 + assert attempts["count"] == 3 - monkeypatch.setattr("tests.support.utils.LangfuseAPI", lambda **_: FakeClient()) - trace = wait_for_trace( - "trace-123", is_result_ready=lambda trace: len(trace["observations"]) == 3 +def test_wait_for_trace_snapshot_waits_for_root_and_scores(monkeypatch): + attempts = {"observations": 0, "scores": 0} + + def get_many(**kwargs): + attempts["observations"] += 1 + observations = [_observation(1, name="child")] + if attempts["observations"] >= 2: + observations.append(_observation(0, name="root", is_root_observation=True)) + return _page(observations) + + def get_many_v3(**kwargs): + attempts["scores"] += 1 + return _page(["score"] if attempts["scores"] >= 3 else []) + + _install_client( + monkeypatch, + observations=SimpleNamespace(get_many=get_many), + scores_v3=SimpleNamespace(get_many_v3=get_many_v3), ) - assert trace["id"] == "trace-123" - assert len(trace["observations"]) == 3 - assert attempts["count"] == 3 + snapshot = wait_for_trace_snapshot( + "trace-123", + min_scores=1, + is_result_ready=lambda trace: trace.root.name == "root", + ) + + assert snapshot.root.name == "root" + assert snapshot.scores == ["score"] + assert attempts["scores"] == 3 + + +def test_trace_snapshot_aggregates_trace_attributes_like_the_platform(): + snapshot = TraceSnapshot( + id="trace-123", + observations=[ + _observation( + 0, + name="root", + trace_name="root", + is_root_observation=True, + input={"q": 1}, + metadata={"root_key": "root", "scope.name": "sdk"}, + tags=["b"], + ), + _observation( + 1, + name="child", + trace_name="explicit-name", + session_id="session-1", + user_id="user-1", + tags=["a"], + public=True, + ), + _observation(2, name="grandchild", session_id="session-2"), + ], + ) + + assert snapshot.name == "explicit-name" + assert snapshot.session_id == "session-2" + assert snapshot.user_id == "user-1" + assert snapshot.tags == ["a", "b"] + assert snapshot.public is True + assert snapshot.input == {"q": 1} + assert snapshot.metadata == {"root_key": "root"} + + +def test_trace_snapshot_name_falls_back_to_root_trace_name(): + snapshot = TraceSnapshot( + id="trace-123", + observations=[ + _observation(0, name="root", trace_name="root", is_root_observation=True), + _observation(1, name="child", trace_name="root"), + ], + ) + + assert snapshot.name == "root" + assert snapshot.session_id is None + assert snapshot.public is False + + +def test_wait_for_scores_follows_cursor_and_polls(monkeypatch): + attempts = {"count": 0} + + def get_many_v3(**kwargs): + attempts["count"] += 1 + if attempts["count"] < 2: + return _page([]) + if kwargs["cursor"] is None: + return _page(["score-1"], cursor="next") + return _page(["score-2"]) + + _install_client(monkeypatch, scores_v3=SimpleNamespace(get_many_v3=get_many_v3)) + + scores = wait_for_scores(min_count=2, trace_id="trace-123") + + assert scores == ["score-1", "score-2"] def test_retry_until_ready_clears_stale_error_after_success(monkeypatch): @@ -171,3 +287,37 @@ def operation(): assert trace["id"] == "trace-123" assert trace["attempt"] == 3 + + +def test_normalize_observation_keeps_scalar_text_io_as_strings(): + observation = normalize_observation( + ObservationV2( + id="obs", + trace_id="trace", + start_time=datetime(2026, 1, 1, tzinfo=timezone.utc), + project_id="project", + type="SPAN", + input="2", + output="true", + ) + ) + + assert observation.input == "2" + assert observation.output == "true" + + +def test_get_observations_keeps_the_latest_row_per_observation_id(monkeypatch): + stale = _observation(0, name="stale", updated_at=START) + fresh = _observation(0, name="fresh", updated_at=START + timedelta(seconds=1)) + + def get_many(**kwargs): + return _page([fresh, stale, _observation(1)]) + + _install_client(monkeypatch, observations=SimpleNamespace(get_many=get_many)) + + observations = get_observations(trace_id="trace-123") + + assert [(o.id, o.name) for o in observations] == [ + ("obs-0", "fresh"), + ("obs-1", None), + ] diff --git a/tests/unit/test_resource_manager.py b/tests/unit/test_resource_manager.py index f66a1e052..f48e61555 100644 --- a/tests/unit/test_resource_manager.py +++ b/tests/unit/test_resource_manager.py @@ -1,5 +1,6 @@ """Test the LangfuseResourceManager and get_client() function.""" +import logging from queue import Queue from types import SimpleNamespace from typing import Sequence @@ -410,6 +411,36 @@ def test_at_fork_reinit_new_httpx_client_uses_configured_timeout_and_headers( client.shutdown() +def test_create_score_with_invalid_input_logs_error_without_enqueueing( + monkeypatch, caplog +): + monkeypatch.setenv("LANGFUSE_MEDIA_UPLOAD_ENABLED", "false") + + with LangfuseResourceManager._lock: + LangfuseResourceManager._instances.clear() + + client = Langfuse( + public_key="pk-invalid-score", + secret_key="sk-invalid-score", + span_exporter=NoOpSpanExporter(), + ) + rm = client._resources + assert rm is not None + enqueued = [] + monkeypatch.setattr(rm, "add_score_task", lambda event, **_: enqueued.append(event)) + + with caplog.at_level(logging.ERROR, logger="langfuse"): + client.create_score(name="invalid", value=object(), trace_id="a" * 32) + + assert enqueued == [] + assert rm._score_ingestion_queue.empty() + error_records = [r for r in caplog.records if r.levelno == logging.ERROR] + assert len(error_records) == 1 + assert "Error creating score" in error_records[0].getMessage() + + client.shutdown() + + def test_stop_and_join_consumer_threads_broadcasts_media_shutdown_after_pausing_all(): events = [] From 8299f526926b6b760233b34058b228c65f324032 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 7 Oct 2026 20:17:50 +0000 Subject: [PATCH 24/25] feat(batch-evaluation)!: evaluate observations only Remove the scope parameter and root-observation mode, so batch evaluation always runs on the observations matching the filter and scores each observation. Resume tokens no longer carry a scope. Co-authored-by: Hassieb Pakzad --- langfuse/_client/client.py | 30 +++---- langfuse/batch_evaluation.py | 132 ++++------------------------ tests/e2e/test_batch_evaluation.py | 51 +++-------- tests/unit/test_batch_evaluation.py | 89 +------------------ 4 files changed, 37 insertions(+), 265 deletions(-) diff --git a/langfuse/_client/client.py b/langfuse/_client/client.py index d6191248a..fe3b09426 100644 --- a/langfuse/_client/client.py +++ b/langfuse/_client/client.py @@ -3153,7 +3153,6 @@ def _create_experiment_run_name( def run_batched_evaluation( self, *, - scope: Literal["observations", "root_observations"], mapper: MapperFunction, filter: Optional[str] = None, fetch_batch_size: int = 50, @@ -3188,11 +3187,6 @@ def run_batched_evaluation( newest first (by start time). Args: - scope: Which observations to evaluate. Must be one of: - - "observations": Every observation matching the filter (spans, - generations, events, ...). Scores are attached to the observation. - - "root_observations": Only the logical root observation of each trace. - Scores are attached to the trace. Use this to evaluate whole traces. mapper: Function that transforms an `ObservationV2` into evaluator inputs. Called as `mapper(item=observation)` and must return an EvaluatorInputs instance with input, output, expected_output, and metadata fields. @@ -3206,7 +3200,7 @@ def run_batched_evaluation( - '[{"type": "arrayOptions", "column": "tags", "operator": "any of", "value": ["production"]}]' - '[{"type": "string", "column": "traceName", "operator": "=", "value": "chat"}]' - '[{"type": "datetime", "column": "startTime", "operator": ">=", "value": "2026-01-01T00:00:00Z"}]' - Default: None (fetches all items of the scope). + Default: None (fetches all observations). fetch_batch_size: Number of items to fetch per API call and hold in memory. Larger values may be faster but use more memory. Maximum 1000. Default: 50. fields: Comma-separated list of observation field groups to fetch. Available @@ -3232,7 +3226,8 @@ def run_batched_evaluation( long-running evaluations. Default: False. resume_from: Optional resume token from a previous run that stopped early (fetch failure or `max_items`). Continues exactly after the last - processed page. Pass the same `scope` and `filter`. Default: None. + processed page. Pass the same `filter`, or omit it to reuse the + token's filter. Default: None. Returns: @@ -3254,17 +3249,17 @@ def run_batched_evaluation( - item_evaluations: Evaluations per observation ID Raises: - ValueError: If an invalid scope or a non-array filter is provided, or the - resume token belongs to a different scope. + ValueError: If a non-array filter is provided, or the resume token was + created for a different filter. Examples: - Evaluate whole traces via their root observations: + Evaluate production observations: ```python from langfuse import Langfuse, EvaluatorInputs, Evaluation client = Langfuse() - def root_mapper(*, item): + def simple_mapper(*, item): return EvaluatorInputs( input=item.input, output=item.output, @@ -3279,15 +3274,14 @@ def length_evaluator(*, input, output, expected_output, metadata): ) result = client.run_batched_evaluation( - scope="root_observations", - mapper=root_mapper, + mapper=simple_mapper, evaluators=[length_evaluator], filter='[{"type": "arrayOptions", "column": "tags", "operator": "any of", "value": ["production"]}]', max_items=1000, verbose=True ) - print(f"Processed {result.total_items_processed} traces") + print(f"Processed {result.total_items_processed} observations") print(f"Created {result.total_scores_created} scores") ``` @@ -3319,7 +3313,6 @@ def composite_evaluator(*, input, output, expected_output, metadata, evaluations return Evaluation(name="composite_score", value=total) result = client.run_batched_evaluation( - scope="observations", mapper=generation_mapper, evaluators=[accuracy_evaluator, relevance_evaluator], composite_evaluator=composite_evaluator, @@ -3331,7 +3324,6 @@ def composite_evaluator(*, input, output, expected_output, metadata, evaluations Continuing a run that stopped early: ```python result = client.run_batched_evaluation( - scope="observations", mapper=generation_mapper, evaluators=[accuracy_evaluator], max_items=10000, @@ -3339,8 +3331,7 @@ def composite_evaluator(*, input, output, expected_output, metadata, evaluations while result.resume_token: result = client.run_batched_evaluation( - scope="observations", - mapper=generation_mapper, + mapper=generation_mapper, evaluators=[accuracy_evaluator], max_items=10000, resume_from=result.resume_token, @@ -3361,7 +3352,6 @@ def composite_evaluator(*, input, output, expected_output, metadata, evaluations BatchEvaluationResult, run_async_safely( runner.run_async( - scope=scope, mapper=mapper, evaluators=evaluators, filter=filter, diff --git a/langfuse/batch_evaluation.py b/langfuse/batch_evaluation.py index 3d299e418..d865a76b0 100644 --- a/langfuse/batch_evaluation.py +++ b/langfuse/batch_evaluation.py @@ -16,13 +16,10 @@ Awaitable, Dict, List, - Literal, Optional, Protocol, - Set, Tuple, Union, - get_args, ) from langfuse.api import ObservationV2 @@ -32,16 +29,6 @@ if TYPE_CHECKING: from langfuse._client.client import Langfuse -BatchEvaluationScope = Literal["observations", "root_observations"] -"""Which observations a batch evaluation runs on. - -- ``"observations"``: every observation matching the filter. Scores are attached - to the observation (``observation_id`` + ``trace_id``). -- ``"root_observations"``: only logical root observations, i.e. one per trace. - Scores are attached to the trace (``trace_id`` only), which makes this the - replacement for trace-level batch evaluation. -""" - DEFAULT_BATCH_EVALUATION_FIELDS = "core,basic,io,metadata" """Default v2 observation field groups fetched for batch evaluation. @@ -71,11 +58,11 @@ class EvaluatorInputs: or any other relevant data that evaluators might use. Examples: - Simple mapper for root observations: + Simple observation mapper: ```python from langfuse import EvaluatorInputs - def root_mapper(*, item): + def simple_mapper(*, item): return EvaluatorInputs( input=item.input, # raw string as returned by the API output=item.output, @@ -161,8 +148,7 @@ def __call__( input/output/expected_output/metadata structure. Args: - item: The `ObservationV2` to transform. With - `scope="root_observations"` this is the root observation of a trace. + item: The `ObservationV2` to transform. Returns: EvaluatorInputs: A structured container with: @@ -175,9 +161,9 @@ def __call__( (for async mappers that need to fetch additional data). Examples: - Basic root observation mapper: + Basic observation mapper: ```python - def map_root(*, item): + def map_basic(*, item): return EvaluatorInputs( input=item.input, output=item.output, @@ -475,8 +461,6 @@ class BatchEvaluationResumeToken: items exist (`has_more_items=True`). Attributes: - scope: The scope of the run ("observations" or "root_observations"). - Resuming requires the same scope. filter: The original JSON filter string used to query items. Pass the same filter when resuming. cursor: Cursor of the next page to fetch. None if no page was fetched yet. @@ -491,7 +475,6 @@ class BatchEvaluationResumeToken: Resuming a run that stopped early: ```python result = client.run_batched_evaluation( - scope="root_observations", mapper=my_mapper, evaluators=[evaluator1, evaluator2], filter=my_filter, @@ -500,7 +483,6 @@ class BatchEvaluationResumeToken: if result.resume_token: result = client.run_batched_evaluation( - scope="root_observations", mapper=my_mapper, evaluators=[evaluator1, evaluator2], filter=my_filter, @@ -527,7 +509,6 @@ class BatchEvaluationResumeToken: def __init__( self, *, - scope: str, filter: Optional[str], last_processed_timestamp: str, last_processed_id: str, @@ -537,7 +518,6 @@ def __init__( """Initialize BatchEvaluationResumeToken with the provided state. Args: - scope: The scope of the run ("observations" or "root_observations"). filter: The original JSON filter string. last_processed_timestamp: ISO 8601 start time of the oldest processed item. last_processed_id: ID of last processed item. @@ -547,7 +527,6 @@ def __init__( Note: All arguments must be provided as keywords. """ - self.scope = scope self.filter = filter self.cursor = cursor self.last_processed_timestamp = last_processed_timestamp @@ -822,7 +801,6 @@ def __init__(self, client: "Langfuse"): async def run_async( self, *, - scope: BatchEvaluationScope, mapper: MapperFunction, evaluators: List[EvaluatorFunction], filter: Optional[str] = None, @@ -845,7 +823,6 @@ async def run_async( statistics. Args: - scope: Which observations to evaluate ("observations", "root_observations"). mapper: Function to transform `ObservationV2` items to evaluator inputs. evaluators: List of evaluation functions to run on each item. filter: JSON filter string (v2 observations filter schema). @@ -856,8 +833,7 @@ async def run_async( composite_evaluator: Optional function to create composite scores. metadata: Metadata to add to all created scores. _add_observation_scores_to_trace: Private option to duplicate - observation-level scores onto the parent trace. Only applies to - scope="observations". + observation-level scores onto the parent trace. max_retries: Maximum retries for failed batch fetches. verbose: If True, log progress to console. resume_from: Resume token from a previous run. If `filter` is omitted, @@ -867,21 +843,11 @@ async def run_async( BatchEvaluationResult with comprehensive statistics. Raises: - ValueError: If the scope is invalid, the filter is not a JSON array, or - the resume token was created for a different scope or filter. + ValueError: If the filter is not a JSON array, or the resume token was + created for a different filter. """ start_time = time.time() - if scope not in get_args(BatchEvaluationScope): - raise ValueError( - f"Invalid scope: {scope!r}. Expected one of " - f"{', '.join(repr(s) for s in get_args(BatchEvaluationScope))}." - ) - if resume_from is not None and resume_from.scope != scope: - raise ValueError( - f"Resume token was created for scope {resume_from.scope!r}, " - f"cannot resume with scope {scope!r}." - ) # The token's cursor only continues the query that produced it. if resume_from is not None: if filter is None: @@ -892,9 +858,7 @@ async def run_async( "same filter, or omit it to reuse the token's filter." ) - effective_filter = self._build_filter( - filter=filter, scope=scope, resume_from=resume_from - ) + effective_filter = self._build_filter(filter=filter, resume_from=resume_from) total_items_fetched = 0 total_items_processed = 0 @@ -931,11 +895,10 @@ async def run_async( and resume_from.last_processed_timestamp else None ) - seen_root_trace_ids: Set[str] = set() batch_number = 0 if verbose: - logger.info("Starting batch evaluation on %s", scope) + logger.info("Starting batch evaluation on observations") if fields: logger.info("Fetching observation fields: %s", fields) if resume_from: @@ -947,7 +910,6 @@ async def run_async( def build_resume_token() -> BatchEvaluationResumeToken: return BatchEvaluationResumeToken( - scope=scope, filter=filter, cursor=cursor, last_processed_timestamp=last_item_timestamp, @@ -1010,7 +972,6 @@ async def process_item( try: result = await self._process_batch_evaluation_item( item=item, - scope=scope, mapper=mapper, evaluators=evaluators, composite_evaluator=composite_evaluator, @@ -1023,10 +984,6 @@ async def process_item( return (item.id, e) items_to_process = [item for item in items if item.id != resumed_item_id] - if scope == "root_observations": - items_to_process = self._select_one_root_per_trace( - items_to_process, seen_root_trace_ids - ) results = await asyncio.gather( *[process_item(item) for item in items_to_process] @@ -1141,7 +1098,6 @@ async def _fetch_batch_with_retry( async def _process_batch_evaluation_item( self, item: ObservationV2, - scope: BatchEvaluationScope, mapper: MapperFunction, evaluators: List[EvaluatorFunction], composite_evaluator: Optional[CompositeEvaluatorFunction], @@ -1153,7 +1109,6 @@ async def _process_batch_evaluation_item( Args: item: The observation to evaluate. - scope: The scope of the run. mapper: Function to transform item to evaluator inputs. evaluators: List of evaluator functions. composite_evaluator: Optional composite evaluator function. @@ -1207,8 +1162,7 @@ async def _process_batch_evaluation_item( ) for evaluation in evaluations: - scores_created += self._create_score_for_scope( - scope=scope, + scores_created += self._create_score( item=item, evaluation=evaluation, additional_metadata=metadata, @@ -1227,8 +1181,7 @@ async def _process_batch_evaluation_item( ) for composite_eval in composite_evals: - composite_scores_created += self._create_score_for_scope( - scope=scope, + composite_scores_created += self._create_score( item=item, evaluation=composite_eval, additional_metadata=metadata, @@ -1343,22 +1296,17 @@ async def _run_composite_evaluator( else: return [] - def _create_score_for_scope( + def _create_score( self, *, - scope: BatchEvaluationScope, item: ObservationV2, evaluation: Evaluation, additional_metadata: Optional[Dict[str, Any]], add_observation_score_to_trace: bool = False, ) -> int: - """Create a score linked to the entity that the scope evaluates. - - `root_observations` scores the trace; `observations` scores the - observation and optionally duplicates the score onto its trace. + """Create a score on the evaluated observation, optionally duplicated onto its trace. Args: - scope: The scope of the run. item: The evaluated observation. evaluation: The evaluation result to create a score from. additional_metadata: Additional metadata to merge with evaluation metadata. @@ -1381,10 +1329,6 @@ def _create_score_for_scope( "config_id": evaluation.config_id, } - if scope == "root_observations": - self.client.create_score(trace_id=item.trace_id, **score_kwargs) - return 1 - self.client.create_score( observation_id=item.id, trace_id=item.trace_id, **score_kwargs ) @@ -1398,10 +1342,9 @@ def _create_score_for_scope( def _build_filter( *, filter: Optional[str], - scope: BatchEvaluationScope, resume_from: Optional[BatchEvaluationResumeToken], ) -> Optional[str]: - """Combine the user filter with the scope and resume constraints. + """Combine the user filter with the resume constraint. Constraints are added as JSON filter conditions rather than query parameters because the API drops a query-parameter filter whenever the @@ -1409,7 +1352,6 @@ def _build_filter( Args: filter: The user-provided JSON filter string (a JSON array). - scope: The scope of the run. resume_from: Optional resume token. Returns: @@ -1430,16 +1372,6 @@ def _build_filter( ) conditions.extend(parsed) - if scope == "root_observations": - conditions.append( - { - "type": "boolean", - "column": "isRootObservation", - "operator": "=", - "value": True, - } - ) - # Results are ordered by start time descending, so items that remain # after the last processed one started before it. The cursor resumes # exactly; the timestamp is the fallback for tokens without a cursor. @@ -1459,40 +1391,6 @@ def _build_filter( return json.dumps(conditions) if conditions else None - @staticmethod - def _select_one_root_per_trace( - items: List[ObservationV2], seen_trace_ids: Set[str] - ) -> List[ObservationV2]: - """Keep one root observation per trace. - - The v2 API marks both physical roots and SDK-detected app roots as root - observations, so a trace can return several. Prefer the physical root - when it is on the same page; otherwise keep the first one seen. - """ - traces_with_physical_root = { - item.trace_id - for item in items - if item.trace_id and not item.parent_observation_id - } - selected: List[ObservationV2] = [] - - for item in items: - if item.trace_id is None: - selected.append(item) - continue - if item.trace_id in seen_trace_ids: - continue - if ( - item.parent_observation_id - and item.trace_id in traces_with_physical_root - ): - continue - - seen_trace_ids.add(item.trace_id) - selected.append(item) - - return selected - def _build_result( self, total_items_fetched: int, diff --git a/tests/e2e/test_batch_evaluation.py b/tests/e2e/test_batch_evaluation.py index 681dab1b5..edbea8ebc 100644 --- a/tests/e2e/test_batch_evaluation.py +++ b/tests/e2e/test_batch_evaluation.py @@ -104,7 +104,7 @@ def length_evaluator(*, output, **kwargs): return Evaluation(name="length", value=float(len(output or ""))) -def test_observations_scope_evaluates_every_observation(corpus): +def test_evaluates_every_observation(corpus): seen: List[ObservationV2] = [] def recording_mapper(*, item): @@ -112,7 +112,6 @@ def recording_mapper(*, item): return io_mapper(item=item) result = get_client().run_batched_evaluation( - scope="observations", mapper=recording_mapper, evaluators=[length_evaluator], filter=corpus.filter, @@ -133,29 +132,6 @@ def recording_mapper(*, item): assert child.output == "child answer 0" -def test_root_observations_scope_scores_traces(corpus): - score_name = f"root-score-{create_uuid()}" - - def root_evaluator(**kwargs): - return Evaluation(name=score_name, value=1.0) - - result = get_client().run_batched_evaluation( - scope="root_observations", - mapper=io_mapper, - evaluators=[root_evaluator], - filter=corpus.filter, - ) - - assert result.completed is True - assert result.total_items_processed == TRACE_COUNT - assert set(result.item_evaluations) == set(corpus.root_ids) - - scores = _wait_for_scores(trace_id=corpus.trace_ids[0], name=score_name) - assert len(scores) == 1 - assert scores[0].subject.kind == "trace" - assert scores[0].subject.id == corpus.trace_ids[0] - - def test_observation_scores_are_attached_to_observations(corpus): score_name = f"obs-score-{create_uuid()}" child_filter = json.loads(corpus.filter) + [ @@ -171,7 +147,6 @@ def observation_evaluator(**kwargs): return Evaluation(name=score_name, value=0.5, comment="ok") result = get_client().run_batched_evaluation( - scope="observations", mapper=io_mapper, evaluators=[observation_evaluator], filter=json.dumps(child_filter), @@ -196,7 +171,6 @@ def observation_evaluator(**kwargs): def test_max_items_then_resume_covers_corpus_exactly_once(corpus): langfuse = get_client() run_kwargs: Any = { - "scope": "observations", "mapper": io_mapper, "evaluators": [length_evaluator], "filter": corpus.filter, @@ -230,7 +204,6 @@ def recording_mapper(*, item): return EvaluatorInputs(input=None, output=None) get_client().run_batched_evaluation( - scope="root_observations", mapper=recording_mapper, evaluators=[length_evaluator], filter=corpus.filter, @@ -240,7 +213,7 @@ def recording_mapper(*, item): assert len(seen) == 1 assert seen[0].input is None - assert seen[0].is_root_observation is True + assert seen[0].output is None def test_composite_evaluator_and_failures(corpus): @@ -251,19 +224,19 @@ def composite(*, evaluations, **kwargs): return Evaluation(name="composite", value=float(len(evaluations))) result = get_client().run_batched_evaluation( - scope="root_observations", mapper=io_mapper, evaluators=[length_evaluator, failing_evaluator], composite_evaluator=composite, filter=corpus.filter, ) - assert result.total_items_processed == TRACE_COUNT - assert result.total_scores_created == TRACE_COUNT - assert result.total_composite_scores_created == TRACE_COUNT - assert result.total_evaluations_failed == TRACE_COUNT + item_count = 2 * TRACE_COUNT + assert result.total_items_processed == item_count + assert result.total_scores_created == item_count + assert result.total_composite_scores_created == item_count + assert result.total_evaluations_failed == item_count stats = {s.name: s for s in result.evaluator_stats} - assert stats["failing_evaluator"].failed_runs == TRACE_COUNT + assert stats["failing_evaluator"].failed_runs == item_count def test_mapper_failures_are_reported_per_item(corpus): @@ -271,21 +244,19 @@ def failing_mapper(*, item): raise ValueError("intentional") result = get_client().run_batched_evaluation( - scope="root_observations", mapper=failing_mapper, evaluators=[length_evaluator], filter=corpus.filter, ) assert result.completed is True - assert result.total_items_failed == TRACE_COUNT - assert set(result.failed_item_ids) == set(corpus.root_ids) - assert result.error_summary == {"ValueError": TRACE_COUNT} + assert result.total_items_failed == 2 * TRACE_COUNT + assert set(result.failed_item_ids) == set(corpus.root_ids + corpus.child_ids) + assert result.error_summary == {"ValueError": 2 * TRACE_COUNT} def test_filter_without_matches_completes_empty(): result = get_client().run_batched_evaluation( - scope="observations", mapper=io_mapper, evaluators=[length_evaluator], filter=_tag_filter(f"nonexistent-{create_uuid()}"), diff --git a/tests/unit/test_batch_evaluation.py b/tests/unit/test_batch_evaluation.py index 8c4240565..4fed9a722 100644 --- a/tests/unit/test_batch_evaluation.py +++ b/tests/unit/test_batch_evaluation.py @@ -103,14 +103,13 @@ def length_evaluator(*, input, output, **kwargs): async def run(runner: BatchEvaluationRunner, **kwargs: Any): - kwargs.setdefault("scope", "observations") kwargs.setdefault("mapper", mapper) kwargs.setdefault("evaluators", [length_evaluator]) return await runner.run_async(**kwargs) @pytest.mark.asyncio -async def test_observations_scope_paginates_with_cursor_and_scores_observations(): +async def test_paginates_with_cursor_and_scores_observations(): runner, api, client = make_runner([make_observation(i) for i in range(5)]) result = await run(runner, fetch_batch_size=2) @@ -141,40 +140,6 @@ async def test_observations_scope_paginates_with_cursor_and_scores_observations( client.flush.assert_called_once() -@pytest.mark.asyncio -async def test_root_observations_scope_filters_roots_and_scores_traces(): - observations = [make_observation(i, is_root=i % 2 == 0) for i in range(6)] - runner, api, client = make_runner(observations) - user_filter = [ - {"type": "string", "column": "traceName", "operator": "=", "value": "chat"} - ] - - result = await run( - runner, - scope="root_observations", - filter=json.dumps(user_filter), - metadata={"run": "nightly"}, - ) - - assert json.loads(api.calls[0]["filter"]) == user_filter + [ - { - "type": "boolean", - "column": "isRootObservation", - "operator": "=", - "value": True, - } - ] - assert set(result.item_evaluations) == {"obs-0", "obs-2", "obs-4"} - scored = [call.kwargs for call in client.create_score.call_args_list] - assert sorted(kwargs["trace_id"] for kwargs in scored) == [ - "trace-0", - "trace-2", - "trace-4", - ] - assert all("observation_id" not in kwargs for kwargs in scored) - assert all(kwargs["metadata"] == {"run": "nightly"} for kwargs in scored) - - @pytest.mark.asyncio async def test_mapper_receives_raw_string_io_and_requested_fields(): runner, api, _ = make_runner([make_observation(1)]) @@ -207,21 +172,8 @@ def test_client_and_runner_share_default_fields(): @pytest.mark.parametrize( ("kwargs", "message"), [ - ({"scope": "traces"}, "Invalid scope"), ({"filter": "not json"}, "JSON array"), ({"filter": '{"tags": ["a"]}'}, "JSON array"), - ( - { - "resume_from": BatchEvaluationResumeToken( - scope="root_observations", - filter=None, - last_processed_timestamp="", - last_processed_id="", - items_processed=0, - ) - }, - "scope", - ), ], ) async def test_invalid_arguments_raise_before_fetching(kwargs, message): @@ -252,7 +204,6 @@ async def test_max_items_aligns_pages_and_resume_continues_without_gaps(): assert token is not None assert token.cursor == "5" assert token.items_processed == 5 - assert token.scope == "observations" second = await run(runner, resume_from=token, fetch_batch_size=3) @@ -304,7 +255,6 @@ async def test_resume_without_cursor_falls_back_to_start_time_bound(): + [make_observation(9, observation_id="obs-tied", start_time=tied_time)] ) token = BatchEvaluationResumeToken( - scope="observations", filter=None, last_processed_timestamp=(BASE_TIME + timedelta(seconds=3)).isoformat(), last_processed_id="obs-3", @@ -332,7 +282,6 @@ async def test_resume_reuses_the_token_filter_and_rejects_a_different_one(): [{"type": "string", "column": "name", "operator": "=", "value": "x"}] ) token = BatchEvaluationResumeToken( - scope="observations", filter=token_filter, cursor="2", last_processed_timestamp=BASE_TIME.isoformat(), @@ -348,42 +297,6 @@ async def test_resume_reuses_the_token_filter_and_rejects_a_different_one(): await run(runner, resume_from=token, filter="[]") -@pytest.mark.asyncio -async def test_root_observations_scope_prefers_the_physical_root_of_a_trace(): - runner, _, client = make_runner( - [ - make_observation(0, trace_id="t1", is_root=True), - # SDK-marked app root below the physical root of the same trace. - make_observation( - 1, trace_id="t1", is_root=True, parent_observation_id="obs-0" - ), - ] - ) - - result = await run(runner, scope="root_observations") - - assert [c.kwargs["trace_id"] for c in client.create_score.call_args_list] == ["t1"] - assert set(result.item_evaluations) == {"obs-0"} - - -@pytest.mark.asyncio -async def test_root_observations_scope_scores_a_trace_once_across_pages(): - # Sibling app roots under a parent that was not exported, on separate pages. - runner, _, client = make_runner( - [ - make_observation( - i, trace_id="t2", is_root=True, parent_observation_id="hidden" - ) - for i in range(3) - ] - ) - - result = await run(runner, scope="root_observations", fetch_batch_size=1) - - assert [c.kwargs["trace_id"] for c in client.create_score.call_args_list] == ["t2"] - assert set(result.item_evaluations) == {"obs-2"} - - @pytest.mark.asyncio async def test_observation_scores_can_be_duplicated_onto_trace(): runner, _, client = make_runner([make_observation(1)]) From 96d5ebb10483555dea8846d52bbd13cc0ad77405 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 7 Oct 2026 20:24:27 +0000 Subject: [PATCH 25/25] fix(batch-evaluation): evaluate duplicate observation rows once The events table can briefly return several rows for one observation, within a page or across a page boundary. Keep the most recently updated row per id so each observation is evaluated and scored once. Also build exception messages before raising. Co-authored-by: Hassieb Pakzad --- langfuse/batch_evaluation.py | 40 +++++++++++++++++++++++++---- tests/unit/test_batch_evaluation.py | 22 ++++++++++++++++ 2 files changed, 57 insertions(+), 5 deletions(-) diff --git a/langfuse/batch_evaluation.py b/langfuse/batch_evaluation.py index d865a76b0..a761f67ce 100644 --- a/langfuse/batch_evaluation.py +++ b/langfuse/batch_evaluation.py @@ -18,6 +18,7 @@ List, Optional, Protocol, + Set, Tuple, Union, ) @@ -895,6 +896,7 @@ async def run_async( and resume_from.last_processed_timestamp else None ) + previous_page_ids: Set[Optional[str]] = set() batch_number = 0 if verbose: @@ -983,7 +985,10 @@ async def process_item( except Exception as e: return (item.id, e) - items_to_process = [item for item in items if item.id != resumed_item_id] + items_to_process = self._deduplicate_page( + items, skip_ids=previous_page_ids | {resumed_item_id} + ) + previous_page_ids = {item.id for item in items} results = await asyncio.gather( *[process_item(item) for item in items_to_process] @@ -1124,7 +1129,8 @@ async def _process_batch_evaluation_item( Exception: If mapping fails or item processing encounters fatal error. """ if not item.trace_id: - raise ValueError(f"Observation {item.id} has no trace_id") + message = f"Observation {item.id} has no trace_id" + raise ValueError(message) scores_created = 0 composite_scores_created = 0 @@ -1365,11 +1371,14 @@ def _build_filter( try: parsed = json.loads(filter) except json.JSONDecodeError as e: - raise ValueError(f"filter must be a JSON array: {e}") from e + message = f"filter must be a JSON array: {e}" + raise ValueError(message) from e if not isinstance(parsed, list): - raise ValueError( - f"filter must be a JSON array of conditions, got {type(parsed).__name__}" + message = ( + "filter must be a JSON array of conditions, " + f"got {type(parsed).__name__}" ) + raise ValueError(message) conditions.extend(parsed) # Results are ordered by start time descending, so items that remain @@ -1391,6 +1400,27 @@ def _build_filter( return json.dumps(conditions) if conditions else None + @staticmethod + def _deduplicate_page( + items: List[ObservationV2], *, skip_ids: Set[Optional[str]] + ) -> List[ObservationV2]: + """Keep one row per observation id, preferring the most recently updated. + + Until ClickHouse merges them, the events table can return several rows + for one observation, within a page or across a page boundary. + """ + latest_by_id: Dict[str, ObservationV2] = {} + for item in items: + if item.id in skip_ids: + continue + current = latest_by_id.get(item.id) + if current is None or (item.updated_at or item.start_time) > ( + current.updated_at or current.start_time + ): + latest_by_id[item.id] = item + + return list(latest_by_id.values()) + def _build_result( self, total_items_fetched: int, diff --git a/tests/unit/test_batch_evaluation.py b/tests/unit/test_batch_evaluation.py index 4fed9a722..1ac1751fa 100644 --- a/tests/unit/test_batch_evaluation.py +++ b/tests/unit/test_batch_evaluation.py @@ -356,3 +356,25 @@ async def async_evaluator(*, output, **kwargs): result = await run(runner, mapper=async_mapper, evaluators=[async_evaluator]) assert result.total_scores_created == 2 + + +@pytest.mark.asyncio +async def test_duplicate_rows_for_one_observation_are_evaluated_once(): + # The events table can briefly return several rows for the same observation. + stale = make_observation(1) + fresh = make_observation(1).model_copy( + update={"output": "fresh answer", "updated_at": BASE_TIME + timedelta(hours=1)} + ) + runner, _, client = make_runner([make_observation(0), stale, fresh]) + seen: List[ObservationV2] = [] + + def recording_mapper(*, item): + seen.append(item) + return mapper(item=item) + + result = await run(runner, mapper=recording_mapper, fetch_batch_size=2) + + assert sorted(item.id for item in seen) == ["obs-0", "obs-1"] + assert [item.output for item in seen if item.id == "obs-1"] == ["fresh answer"] + assert result.total_items_processed == 2 + assert client.create_score.call_count == 2