From 25cd3c0b578e7c0fbd4eb3e91e8052abd833f441 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 7 Oct 2026 21:24:18 +0000 Subject: [PATCH 1/2] chore(api): drop references to API types removed from the v4 spec langfuse.model no longer re-exports DatasetRun, Observation and TraceWithFullDetails, which the regenerated API client removes. The api docstrings point to the v2 observations, scores_v3 and experiments reads instead of trace.get/list and the legacy v1 resources. Two unit tests no longer mock the removed dataset_run_items resource. Co-authored-by: Hassieb Pakzad --- langfuse/_client/client.py | 35 ++++++++++--------------- langfuse/model.py | 3 --- tests/unit/test_experiment.py | 7 +---- tests/unit/test_propagate_attributes.py | 8 ------ 4 files changed, 15 insertions(+), 38 deletions(-) diff --git a/langfuse/_client/client.py b/langfuse/_client/client.py index a588b0e99..5889bd35c 100644 --- a/langfuse/_client/client.py +++ b/langfuse/_client/client.py @@ -426,23 +426,17 @@ def api(self) -> LangfuseAPI: Semantics that are easy to miss: - **Ingestion is asynchronous.** `langfuse.flush()` only guarantees delivery to - the API, not read visibility: reads such as `api.trace.get(trace_id)` may - raise `langfuse.api.NotFoundError` until processing completes (typically - within 15-30 seconds; longer under load). The same applies to scores and - dataset run reads. Instead of a fixed sleep, retry with a deadline: - - - **List endpoints return lightweight views.** `api.trace.list(...)` returns - `TraceWithDetails`, where `observations` and `scores` are lists of ID strings. - Fetch the full objects with `api.trace.get(trace_id)` (`TraceWithFullDetails`), - or prefer `api.observations.get_many(trace_id=...)` for row-level observation - queries. The same list-view vs. get-detail pattern applies to other resources. - - - **Prefer the v2 data APIs — they are the defaults since SDK v4.** - `api.observations` and `api.metrics` map to the high-performance - `/api/public/v2/...` endpoints and are the recommended read path. Their v1 - equivalents remain available under `api.legacy.observations_v1` / - `api.legacy.metrics_v1` but are less performant at scale, not recommended - for new workflows, and will be deprecated. + the API, not read visibility: reads such as + `api.observations.get_many(trace_id=...)` may return no data until + processing completes (typically within 15-30 seconds; longer under load). + The same applies to scores and experiment reads. Instead of a fixed sleep, + poll with a deadline (see the ingestion-lag link below). + + - **Observations are the read model.** Read trace data with + `api.observations.get_many(trace_id=...)` (`/api/public/v2/observations`, + cursor-paginated; request field groups with `fields`). A trace's name, + user, session, tags and input/output are on its root observation. Read + scores with `api.scores_v3` and experiments with `api.experiments`. - For large-scale aggregation (usage/cost by model, user, etc.), prefer the v2 Metrics API (`api.metrics.metrics(...)`) over paginating row-level data. @@ -450,7 +444,7 @@ def api(self) -> LangfuseAPI: See also: `async_api`, https://langfuse.com/docs/api-and-data-platform/features/query-via-sdk - (ingestion lag: #ingestion-lag, list vs. get: #traces-list-vs-get), + (ingestion lag: #ingestion-lag), https://langfuse.com/docs/api-and-data-platform/features/observations-api, https://langfuse.com/docs/metrics/features/metrics-api """ @@ -2247,9 +2241,8 @@ def flush(self) -> None: `flush()` guarantees data was *delivered* to the API, not that it is *readable* yet: server-side ingestion is asynchronous, so flushed data may not be queryable for 15-30 seconds — - `api.observations.get_many(trace_id=...)` may return empty results and - `api.trace.get()` may raise `langfuse.api.NotFoundError` right after a - successful flush. See the `api` property docs for a bounded retry + `api.observations.get_many(trace_id=...)` may return empty results right + after a successful flush. See the `api` property docs for a bounded retry pattern, or https://langfuse.com/docs/api-and-data-platform/features/query-via-sdk#ingestion-lag """ diff --git a/langfuse/model.py b/langfuse/model.py index 69d721597..30c6fe0f3 100644 --- a/langfuse/model.py +++ b/langfuse/model.py @@ -9,14 +9,11 @@ CreateDatasetRequest, # noqa Dataset, # noqa DatasetItem, # noqa - DatasetRun, # noqa DatasetStatus, # noqa MapValue, # noqa - Observation, # noqa Prompt, Prompt_Chat, Prompt_Text, - TraceWithFullDetails, # noqa ) from langfuse.logger import langfuse_logger diff --git a/tests/unit/test_experiment.py b/tests/unit/test_experiment.py index 02596c242..43d0856db 100644 --- a/tests/unit/test_experiment.py +++ b/tests/unit/test_experiment.py @@ -598,15 +598,11 @@ def test_dataset_experiment_without_project_id_uses_random_id( assert all(len(i) == 16 for i in ids) assert "random experiment id" in caplog.text - def test_dataset_run_uses_stable_id_url_version_and_no_run_item_post( + def test_dataset_run_uses_stable_id_url_and_version( self, langfuse_memory_client, find_spans, monkeypatch ): create_score = MagicMock() - create_run_item = MagicMock() monkeypatch.setattr(langfuse_memory_client, "create_score", create_score) - monkeypatch.setattr( - langfuse_memory_client.api.dataset_run_items, "create", create_run_item - ) monkeypatch.setattr(langfuse_memory_client, "_get_project_id", lambda: "p") version = datetime(2026, 2, 3, 4, 5, 6, tzinfo=timezone.utc) @@ -623,7 +619,6 @@ def test_dataset_run_uses_stable_id_url_version_and_no_run_item_post( ) langfuse_memory_client.flush() - create_run_item.assert_not_called() assert result.experiment_id == "eab007015b1c6f77" assert ( result.experiment_url diff --git a/tests/unit/test_propagate_attributes.py b/tests/unit/test_propagate_attributes.py index 77afcd7cf..b16a89844 100644 --- a/tests/unit/test_propagate_attributes.py +++ b/tests/unit/test_propagate_attributes.py @@ -2873,14 +2873,6 @@ def test_experiment_attributes_propagate_with_dataset( ): """Test experiment attribute propagation with Langfuse dataset.""" - def fail_create_dataset_run_item(*args, **kwargs): - raise AssertionError("run_experiment must not create dataset run items") - - monkeypatch.setattr( - langfuse_client.api.dataset_run_items, - "create", - fail_create_dataset_run_item, - ) monkeypatch.setattr( langfuse_client, "_get_project_id", lambda: "test-project-id" ) From 3b806e0b2e008e9149d61579b96860f1d80e3fec Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 7 Oct 2026 21:30:18 +0000 Subject: [PATCH 2/2] chore: re-run CI Co-authored-by: Hassieb Pakzad