Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
35 changes: 14 additions & 21 deletions langfuse/_client/client.py
Original file line number Diff line number Diff line change
Expand Up @@ -426,31 +426,25 @@ def api(self) -> LangfuseAPI:
Semantics that are easy to miss:

- **Ingestion is asynchronous.** `langfuse.flush()` only guarantees delivery to
the API, not read visibility: reads such as `api.trace.get(trace_id)` may
raise `langfuse.api.NotFoundError` until processing completes (typically
within 15-30 seconds; longer under load). The same applies to scores and
dataset run reads. Instead of a fixed sleep, retry with a deadline:

- **List endpoints return lightweight views.** `api.trace.list(...)` returns
`TraceWithDetails`, where `observations` and `scores` are lists of ID strings.
Fetch the full objects with `api.trace.get(trace_id)` (`TraceWithFullDetails`),
or prefer `api.observations.get_many(trace_id=...)` for row-level observation
queries. The same list-view vs. get-detail pattern applies to other resources.

- **Prefer the v2 data APIs — they are the defaults since SDK v4.**
`api.observations` and `api.metrics` map to the high-performance
`/api/public/v2/...` endpoints and are the recommended read path. Their v1
equivalents remain available under `api.legacy.observations_v1` /
`api.legacy.metrics_v1` but are less performant at scale, not recommended
for new workflows, and will be deprecated.
the API, not read visibility: reads such as
`api.observations.get_many(trace_id=...)` may return no data until
processing completes (typically within 15-30 seconds; longer under load).
The same applies to scores and experiment reads. Instead of a fixed sleep,
poll with a deadline (see the ingestion-lag link below).

- **Observations are the read model.** Read trace data with
`api.observations.get_many(trace_id=...)` (`/api/public/v2/observations`,
cursor-paginated; request field groups with `fields`). A trace's name,
user, session, tags and input/output are on its root observation. Read
scores with `api.scores_v3` and experiments with `api.experiments`.

- For large-scale aggregation (usage/cost by model, user, etc.), prefer the
v2 Metrics API (`api.metrics.metrics(...)`) over paginating row-level data.


See also: `async_api`,
https://langfuse.com/docs/api-and-data-platform/features/query-via-sdk
(ingestion lag: #ingestion-lag, list vs. get: #traces-list-vs-get),
(ingestion lag: #ingestion-lag),
https://langfuse.com/docs/api-and-data-platform/features/observations-api,
https://langfuse.com/docs/metrics/features/metrics-api
"""
Expand Down Expand Up @@ -2247,9 +2241,8 @@ def flush(self) -> None:
`flush()` guarantees data was *delivered* to the API, not that it is
*readable* yet: server-side ingestion is asynchronous, so flushed data
may not be queryable for 15-30 seconds —
`api.observations.get_many(trace_id=...)` may return empty results and
`api.trace.get()` may raise `langfuse.api.NotFoundError` right after a
successful flush. See the `api` property docs for a bounded retry
`api.observations.get_many(trace_id=...)` may return empty results right
after a successful flush. See the `api` property docs for a bounded retry
pattern, or
https://langfuse.com/docs/api-and-data-platform/features/query-via-sdk#ingestion-lag
"""
Expand Down
3 changes: 0 additions & 3 deletions langfuse/model.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,14 +9,11 @@
CreateDatasetRequest, # noqa
Dataset, # noqa
DatasetItem, # noqa
DatasetRun, # noqa
DatasetStatus, # noqa
MapValue, # noqa
Observation, # noqa
Prompt,
Prompt_Chat,
Prompt_Text,
TraceWithFullDetails, # noqa
)
from langfuse.logger import langfuse_logger

Expand Down
7 changes: 1 addition & 6 deletions tests/unit/test_experiment.py
Original file line number Diff line number Diff line change
Expand Up @@ -598,15 +598,11 @@ def test_dataset_experiment_without_project_id_uses_random_id(
assert all(len(i) == 16 for i in ids)
assert "random experiment id" in caplog.text

def test_dataset_run_uses_stable_id_url_version_and_no_run_item_post(
def test_dataset_run_uses_stable_id_url_and_version(
self, langfuse_memory_client, find_spans, monkeypatch
):
create_score = MagicMock()
create_run_item = MagicMock()
monkeypatch.setattr(langfuse_memory_client, "create_score", create_score)
monkeypatch.setattr(
langfuse_memory_client.api.dataset_run_items, "create", create_run_item
)
monkeypatch.setattr(langfuse_memory_client, "_get_project_id", lambda: "p")
version = datetime(2026, 2, 3, 4, 5, 6, tzinfo=timezone.utc)

Expand All @@ -623,7 +619,6 @@ def test_dataset_run_uses_stable_id_url_version_and_no_run_item_post(
)
langfuse_memory_client.flush()

create_run_item.assert_not_called()
assert result.experiment_id == "eab007015b1c6f77"
assert (
result.experiment_url
Expand Down
8 changes: 0 additions & 8 deletions tests/unit/test_propagate_attributes.py
Original file line number Diff line number Diff line change
Expand Up @@ -2873,14 +2873,6 @@ def test_experiment_attributes_propagate_with_dataset(
):
"""Test experiment attribute propagation with Langfuse dataset."""

def fail_create_dataset_run_item(*args, **kwargs):
raise AssertionError("run_experiment must not create dataset run items")

monkeypatch.setattr(
langfuse_client.api.dataset_run_items,
"create",
fail_create_dataset_run_item,
)
monkeypatch.setattr(
langfuse_client, "_get_project_id", lambda: "test-project-id"
)
Expand Down
Loading