Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/workflows/codeql.yml
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,7 @@ on:
pull_request:
branches: [ main ]
schedule:
- cron: '43 4 * * *' # Daily at 04:43 UTC.
- cron: '43 20 * * 0' # Sundays at 20:43 UTC.

permissions:
contents: read
Expand Down
3 changes: 0 additions & 3 deletions AGENTS.md
Original file line number Diff line number Diff line change
Expand Up @@ -166,9 +166,6 @@ VCR_MODE=record ./gradlew :braintrust-sdk:test --tests 'dev.braintrust.devserver
- **`braintrust-api` is generated code.** don't edit sources under it by hand; it's regenerated from the braintrust openapi spec pinned as `braintrustOpenApiRef` in gradle.properties.
- **there are no version constants to bump.** the sdk version is derived from git tags at build time (`generateVersion()` in build.gradle) and written into braintrust.properties. "bump the version" is not a source change.
- When adding test cases, favor adding to the test file of the module being changed rather than making a new file. For example, if you fix a bug in the `Foo` module, add the test case to `FooTest.java` instead of making a new file, `FooTestMyBuggyCase.java`
- **run CodeQL once as a final check for code changes:** `./gradlew checkCodeQL` (first-time setup: `mise trust && mise install`). don't rerun it after every edit; rerun when needed to verify a security fix. the task fails on any finding or scan error; review the printed findings and generated SARIF report.
- **use judgment when resolving CodeQL findings.** investigate the flagged code path and fix genuine vulnerabilities at the source. don't suppress alerts, weaken checks, or distort correct code just to make findings disappear. if a finding appears to be a false positive or there is a good reason not to follow its recommendation, tell the user which finding, the evidence and security tradeoff, and your proposed disposition. get their agreement before suppressing or dismissing it; don't silently ignore it.
- **run dependency checker when changing dependencies:**: `./gradlew checkDependencies`
- **don't add new build tools without approval** favor using java/gradle/groovy/bash for misc scripts and tools. Favor using the existing ecosystem instead of introducing new build dependencies. If you think a new build dependency is worth it, ask for approval to add it.

## Releasing
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,7 @@

import com.anthropic.core.RequestOptions;
import com.anthropic.core.http.HttpClient;
import com.anthropic.core.http.HttpMethod;
import com.anthropic.core.http.HttpRequest;
import com.anthropic.core.http.HttpRequestBody;
import com.anthropic.core.http.HttpResponse;
Expand All @@ -20,6 +21,7 @@
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import java.util.Set;
import java.util.concurrent.CompletableFuture;
import java.util.concurrent.atomic.AtomicBoolean;
import java.util.concurrent.atomic.AtomicLong;
Expand Down Expand Up @@ -50,7 +52,7 @@ TracingHttpClient withTracer(Tracer tracer) {
}

/**
* Starts the LLM span. anthropic-java (and frameworks like Spring AI 2.x) dispatch
* Starts a request span. anthropic-java (and frameworks like Spring AI 2.x) dispatch
* async/streaming requests on executors where the caller's thread-local context is lost — which
* would orphan the span. {@link ContextCapturingProxy} captures the caller's context at the
* service-call boundary and threads it through as {@code headerContext}; when that is absent we
Expand All @@ -59,11 +61,9 @@ TracingHttpClient withTracer(Tracer tracer) {
* instrumented — a long-lived client wrapped inside some unrelated span would otherwise parent
* every future request to that stale span.
*/
private Span startLlmSpan(@Nullable Context headerContext) {
private Span startSpan(String name, @Nullable Context headerContext) {
Context parent = headerContext != null ? headerContext : Context.current();
return tracer.spanBuilder(InstrumentationSemConv.UNSET_LLM_SPAN_NAME)
.setParent(parent)
.startSpan();
return tracer.spanBuilder(name).setParent(parent).startSpan();
}

/**
Expand Down Expand Up @@ -118,7 +118,10 @@ public void close() {
public @Nonnull HttpResponse execute(
@Nonnull HttpRequest httpRequest, @Nonnull RequestOptions requestOptions) {
var extracted = extractCallerContext(httpRequest);
var span = startLlmSpan(extracted.callerContext());
if (!isLlmRequest(extracted.request())) {
return executeHttp(extracted, requestOptions);
}
var span = startSpan(InstrumentationSemConv.UNSET_LLM_SPAN_NAME, extracted.callerContext());
try (var ignored = span.makeCurrent()) {
var bufferedRequest = bufferRequestBody(extracted.request());

Expand Down Expand Up @@ -150,7 +153,10 @@ public void close() {
public @Nonnull CompletableFuture<HttpResponse> executeAsync(
@Nonnull HttpRequest httpRequest, @Nonnull RequestOptions requestOptions) {
var extracted = extractCallerContext(httpRequest);
var span = startLlmSpan(extracted.callerContext());
if (!isLlmRequest(extracted.request())) {
return executeHttpAsync(extracted, requestOptions);
}
var span = startSpan(InstrumentationSemConv.UNSET_LLM_SPAN_NAME, extracted.callerContext());
try {
var bufferedRequest = bufferRequestBody(extracted.request());
String inputJson =
Expand Down Expand Up @@ -187,6 +193,63 @@ public void close() {
}
}

/**
* Only {@code POST .../messages} and the legacy {@code POST .../complete} produce model output
* and get the full LLM treatment. Everything else ({@code messages/batches}, {@code
* messages/count_tokens}, models, files, ...) gets a plain {@code anthropic.http} span with the
* body left untouched.
*/
private static final Set<String> LLM_ENDPOINTS = Set.of("messages", "complete");

private static boolean isLlmRequest(HttpRequest request) {
var path = request.pathSegments();
return request.method() == HttpMethod.POST
&& !path.isEmpty()
&& LLM_ENDPOINTS.contains(path.get(path.size() - 1));
}

private HttpResponse executeHttp(ExtractedRequest extracted, RequestOptions requestOptions) {
var span = startSpan("anthropic.http", extracted.callerContext());
try (var ignored = span.makeCurrent()) {
var response = underlying.execute(extracted.request(), requestOptions);
InstrumentationSemConv.tagHttpSpanResponse(span, response.statusCode());
return response;
} catch (Exception e) {
InstrumentationSemConv.tagLLMSpanResponse(span, e);
throw e;
} finally {
span.end();
}
}

private CompletableFuture<HttpResponse> executeHttpAsync(
ExtractedRequest extracted, RequestOptions requestOptions) {
var span = startSpan("anthropic.http", extracted.callerContext());
try (var ignored = span.makeCurrent()) {
return underlying
.executeAsync(extracted.request(), requestOptions)
.whenComplete(
(response, error) -> {
try {
if (error != null) {
InstrumentationSemConv.tagLLMSpanResponse(span, error);
} else {
InstrumentationSemConv.tagHttpSpanResponse(
span, response.statusCode());
}
} finally {
span.end();
}
})
// Isolate cleanup from cancellation of the caller's future.
.copy();
} catch (Exception e) {
InstrumentationSemConv.tagLLMSpanResponse(span, e);
span.end();
throw e;
}
}

// -------------------------------------------------------------------------
// Request buffering — identical pattern to OpenAI TracingHttpClient
// -------------------------------------------------------------------------
Expand Down
Loading
Loading