From 7b0fbefd1f4a6b370e4200110477f1f395ddb834 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 04:54:24 +0000 Subject: [PATCH 1/9] refactor(stream)!: remove unused StreamingToolCallTracker No crate in the workspace referenced the tracker; providers accumulate streamed tool-call deltas themselves. Drops the module, its 787-line test file and the re-exported companion types. Breaking for aimux-stream's public Rust API; noted in CHANGELOG under Unreleased. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_015EYDYDWYcPsDmjBjFeuDVe --- CHANGELOG.md | 11 + aimux-stream/src/lib.rs | 5 - .../src/streaming_tool_call_tracker.rs | 429 ---------- .../tests/streaming_tool_call_tracker_test.rs | 787 ------------------ 4 files changed, 11 insertions(+), 1221 deletions(-) delete mode 100644 aimux-stream/src/streaming_tool_call_tracker.rs delete mode 100644 aimux-stream/tests/streaming_tool_call_tracker_test.rs diff --git a/CHANGELOG.md b/CHANGELOG.md index 84c46819..a8b303ad 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,6 +5,17 @@ All notable changes to aimux are documented in this file. The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). +## [Unreleased] + +### Breaking + +**Rust (aimux-stream)** + +- Removed `StreamingToolCallTracker` and its companion types + (`StreamingToolCallDelta`, `StreamingToolCallFunction`, `ToolCallStreamPart`, + `TrackerError`, `TypeValidation`). No crate in the workspace used them: + providers accumulate streamed tool-call deltas themselves. + ## [0.5.0] - 2026-09-27 **Breaking release.** 13 PRs since 0.3.0: the cross-language error model diff --git a/aimux-stream/src/lib.rs b/aimux-stream/src/lib.rs index 8091dbeb..19d442c5 100644 --- a/aimux-stream/src/lib.rs +++ b/aimux-stream/src/lib.rs @@ -6,13 +6,8 @@ pub mod lines; pub mod ndjson; pub mod sse; -pub mod streaming_tool_call_tracker; // Re-export the most commonly used items. pub use lines::extract_lines; pub use ndjson::{NdjsonError, NdjsonStream}; pub use sse::{SseError, SseEvent, SseStream}; -pub use streaming_tool_call_tracker::{ - StreamingToolCallDelta, StreamingToolCallFunction, StreamingToolCallTracker, - ToolCallStreamPart, TrackerError, TypeValidation, -}; diff --git a/aimux-stream/src/streaming_tool_call_tracker.rs b/aimux-stream/src/streaming_tool_call_tracker.rs deleted file mode 100644 index be9ad890..00000000 --- a/aimux-stream/src/streaming_tool_call_tracker.rs +++ /dev/null @@ -1,429 +0,0 @@ -//! Streaming tool call tracker. -//! -//! Rust translation of `@ai-sdk/provider-utils`'s `StreamingToolCallTracker` -//! (`packages/provider-utils/src/streaming-tool-call-tracker.ts`). -//! -//! Tracks streaming tool call state across multiple deltas from an -//! OpenAI-compatible chat completion stream: accumulates `arguments` string -//! fragments by `index`, emits `tool-input-start` / `tool-input-delta` / -//! `tool-input-end` / `tool-call` events, and finalizes any unfinished tool -//! calls on [`StreamingToolCallTracker::flush`]. -//! -//! Like the TS original, a tool call is *never* finalized before `flush` — a -//! parsable argument buffer can still be the prefix of a longer argument -//! string, so acting on it early would use truncated inputs (ai-sdk #13137). - -use serde_json::Value; -use thiserror::Error; - -/// Default upper bound accepted for a tool call `index`. -const DEFAULT_MAX_INDEX: usize = 1024; - -/// The `function` sub-object of a streaming tool call delta. -#[derive(Debug, Clone, Default)] -pub struct StreamingToolCallFunction { - pub name: Option, - pub arguments: Option, -} - -/// A streaming tool call delta — the `tool_calls[i]` entry of an OpenAI-style -/// streaming chunk. -/// -/// Use the builder methods ([`StreamingToolCallDelta::index`], etc.) to -/// construct one ergonomically. `arguments: null` (TS) maps to `None`; an -/// empty-string `arguments: ''` maps to `Some("")`. -#[derive(Debug, Clone, Default)] -pub struct StreamingToolCallDelta { - pub index: Option, - pub id: Option, - /// The `type` field. Named `r#type` because `type` is a reserved word. - pub r#type: Option, - pub function: Option, - /// Provider-specific payload carried alongside the standard fields. Used by - /// `extract_metadata` to pull out provider metadata (e.g. a Google thought - /// signature). Defaults to [`Value::Null`]. - pub extra: Value, -} - -impl StreamingToolCallDelta { - #[must_use] - pub fn new() -> Self { - Self::default() - } - - #[must_use] - pub fn index(mut self, index: usize) -> Self { - self.index = Some(index); - self - } - - #[must_use] - pub fn id(mut self, id: impl Into) -> Self { - self.id = Some(id.into()); - self - } - - /// Set the `type` field (named `tool_type` because `type` is reserved). - #[must_use] - pub fn tool_type(mut self, t: impl Into) -> Self { - self.r#type = Some(t.into()); - self - } - - #[must_use] - pub fn function_name(mut self, name: impl Into) -> Self { - self.function.get_or_insert_with(Default::default).name = Some(name.into()); - self - } - - /// Set the `function.arguments` fragment. Pass `""` for an explicit empty - /// fragment; omit the call entirely for `None` (TS `arguments: null`). - #[must_use] - pub fn arguments(mut self, args: impl Into) -> Self { - self.function.get_or_insert_with(Default::default).arguments = Some(args.into()); - self - } - - #[must_use] - pub fn extra(mut self, extra: Value) -> Self { - self.extra = extra; - self - } -} - -/// The stream parts emitted by [`StreamingToolCallTracker`]. -/// -/// Mirrors the subset of `LanguageModelV4StreamPart` the TS tracker enqueues. -/// Note that [`ToolCallStreamPart::ToolCall::input`] is the raw accumulated -/// argument *string* — the tracker does not parse it (matching the TS -/// behavior). -#[derive(Debug, Clone, PartialEq)] -pub enum ToolCallStreamPart { - /// Start of a tool call's input streaming. - ToolInputStart { id: String, tool_name: String }, - /// A delta of tool call input (a partial argument fragment). - ToolInputDelta { id: String, delta: String }, - /// End of a tool call's input streaming. - ToolInputEnd { id: String }, - /// A complete, finalized tool call. - ToolCall { - tool_call_id: String, - tool_name: String, - input: String, - provider_metadata: Option, - }, -} - -/// How to validate the `type` field on a new tool call delta. -#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] -pub enum TypeValidation { - /// No validation (default). - #[default] - None, - /// Throw if `type` is present and not `"function"`. - IfPresent, - /// Throw if `type` is not exactly `"function"`. - Required, -} - -/// Errors raised while processing a tool call delta. -#[derive(Debug, Error, PartialEq)] -pub enum TrackerError { - #[error("Expected 'id' to be a string.")] - MissingId, - #[error("Expected 'function.name' to be a string.")] - MissingFunctionName, - #[error("Expected 'function' type.")] - InvalidType, - #[error("Tool call index out of range")] - IndexOutOfRange, -} - -struct TrackedToolCall { - id: String, - function_name: String, - arguments: String, - has_finished: bool, - metadata: Option, -} - -/// Extract provider metadata from a delta (the TS `extractMetadata` option). -type ExtractMetadataFn = Box Option>; -/// Build the `providerMetadata` for a finalized tool call (the TS -/// `buildToolCallProviderMetadata` option). -type BuildMetadataFn = Box) -> Option>; - -/// Tracks streaming tool call state across multiple deltas from an -/// OpenAI-compatible chat completion stream. -/// -/// Emitted [`ToolCallStreamPart`]s accumulate in an internal buffer; inspect -/// them with [`parts`](Self::parts) and reset between checks with -/// [`clear_parts`](Self::clear_parts) (the TS test's `parts.length = 0`). -/// -/// The type parameter `M` is the provider-metadata type. Use `()` (the -/// default) when no metadata handling is needed, or `serde_json::Value` (or -/// any `SharedV4ProviderMetadata`-like type) together with -/// [`with_extract_metadata`](Self::with_extract_metadata) / -/// [`with_build_provider_metadata`](Self::with_build_provider_metadata). -pub struct StreamingToolCallTracker { - tool_calls: Vec>>, - parts: Vec>, - // The TS `generateId` option: fallback for `toolCall.id ?? generateId()` - // when an incoming tool-call delta omits its id. - generate_id: Box String>, - type_validation: TypeValidation, - /// Upper bound accepted for a tool call `index`. Guards against a remote - /// index resizing `tool_calls` to a huge vector. Defaults to - /// [`DEFAULT_MAX_INDEX`] (1024). - max_index: usize, - extract_metadata: Option>, - build_provider_metadata: Option>, -} - -impl Default for StreamingToolCallTracker<()> { - fn default() -> Self { - Self::new() - } -} - -impl StreamingToolCallTracker { - /// Create a new tracker with no metadata handling and default settings. - #[must_use] - pub fn new() -> Self { - Self { - tool_calls: Vec::new(), - parts: Vec::new(), - generate_id: Box::new(|| "generated".to_string()), - type_validation: TypeValidation::None, - max_index: DEFAULT_MAX_INDEX, - extract_metadata: None, - build_provider_metadata: None, - } - } - - /// Set a custom id generator (the TS `generateId` option). - #[must_use] - pub fn with_generate_id String + 'static>(mut self, f: F) -> Self { - self.generate_id = Box::new(f); - self - } - - /// Set the `type` validation mode (the TS `typeValidation` option). - #[must_use] - pub fn with_type_validation(mut self, v: TypeValidation) -> Self { - self.type_validation = v; - self - } - - /// Set the maximum accepted tool call `index` (defaults to 1024). A delta - /// whose resolved `index` exceeds `max_index` returns - /// [`TrackerError::IndexOutOfRange`] instead of resizing `tool_calls` to - /// `index + 1` slots. - #[must_use] - pub fn with_max_index(mut self, max_index: usize) -> Self { - self.max_index = max_index; - self - } - - /// Set the metadata extractor (the TS `extractMetadata` option). Called - /// once when a new tool call is detected; the returned metadata is stored - /// on the tool call and passed to the builder at finalization. - #[must_use] - pub fn with_extract_metadata Option + 'static>( - mut self, - f: F, - ) -> Self { - self.extract_metadata = Some(Box::new(f)); - self - } - - /// Set the provider-metadata builder (the TS `buildToolCallProviderMetadata` - /// option). Receives the metadata previously extracted; if `None` is - /// returned, no `provider_metadata` is included in the `tool-call` event. - #[must_use] - pub fn with_build_provider_metadata) -> Option + 'static>( - mut self, - f: F, - ) -> Self { - self.build_provider_metadata = Some(Box::new(f)); - self - } - - /// Process a tool call delta from a streaming response chunk. Emits events - /// into the internal buffer. - /// - /// # Errors - /// - /// Returns `TrackerError::IndexOutOfRange` when the delta's tool index - /// exceeds the configured maximum. - pub fn process_delta(&mut self, delta: &StreamingToolCallDelta) -> Result<(), TrackerError> { - let index = delta.index.unwrap_or(self.tool_calls.len()); - // Guard against a remote `index` resizing `tool_calls` to a huge vector. - if index > self.max_index { - return Err(TrackerError::IndexOutOfRange); - } - let is_new = self - .tool_calls - .get(index) - .map(std::option::Option::is_none) - .unwrap_or(true); - if is_new { - self.process_new_tool_call(index, delta)?; - } else { - self.process_existing_tool_call(index, delta); - } - Ok(()) - } - - /// Finalize any unfinished tool calls. Should be called during the - /// stream's flush to ensure all tool calls are properly completed. Emits - /// `tool-input-end` and `tool-call` for each unfinished tool call. - pub fn flush(&mut self) { - for i in 0..self.tool_calls.len() { - let needs_finish = self - .tool_calls - .get(i) - .and_then(|o| o.as_ref()) - .is_some_and(|tc| !tc.has_finished); - if needs_finish { - self.finish_tool_call_at(i); - } - } - } - - /// The events emitted so far (accumulated across `process_delta`/`flush`). - #[must_use] - pub fn parts(&self) -> &[ToolCallStreamPart] { - &self.parts - } - - /// Clear the accumulated events (the TS test's `parts.length = 0`). - pub fn clear_parts(&mut self) { - self.parts.clear(); - } - - fn process_new_tool_call( - &mut self, - index: usize, - delta: &StreamingToolCallDelta, - ) -> Result<(), TrackerError> { - match self.type_validation { - TypeValidation::Required => { - if delta.r#type.as_deref() != Some("function") { - return Err(TrackerError::InvalidType); - } - } - TypeValidation::IfPresent => { - if delta.r#type.as_deref().is_some_and(|t| t != "function") { - return Err(TrackerError::InvalidType); - } - } - TypeValidation::None => {} - } - - let id = delta.id.clone().unwrap_or_else(|| (self.generate_id)()); - let function_name = delta - .function - .as_ref() - .and_then(|f| f.name.clone()) - .ok_or(TrackerError::MissingFunctionName)?; - - self.parts.push(ToolCallStreamPart::ToolInputStart { - id: id.clone(), - tool_name: function_name.clone(), - }); - - let metadata = self - .extract_metadata - .as_ref() - .and_then(|extract| extract(delta)); - - // TS: `toolCallDelta.function.arguments ?? ''`. - let arguments = delta - .function - .as_ref() - .and_then(|f| f.arguments.clone()) - .unwrap_or_default(); - - if index >= self.tool_calls.len() { - self.tool_calls.resize_with(index + 1, || None); - } - self.tool_calls[index] = Some(TrackedToolCall { - id: id.clone(), - function_name: function_name.clone(), - arguments: arguments.clone(), - has_finished: false, - metadata, - }); - - // Emit initial delta if arguments already present. - if !arguments.is_empty() { - self.parts.push(ToolCallStreamPart::ToolInputDelta { - id: id.clone(), - delta: arguments, - }); - } - - // Tool calls must not finalize before the stream ends (see #13137). - Ok(()) - } - - fn process_existing_tool_call(&mut self, index: usize, delta: &StreamingToolCallDelta) { - // TS: `toolCallDelta.function?.arguments != null`. - let new_args = match delta.function.as_ref().and_then(|f| f.arguments.as_ref()) { - Some(args) => args.clone(), - None => return, - }; - let id = { - let Some(Some(tool_call)) = self.tool_calls.get_mut(index) else { - return; - }; - if tool_call.has_finished { - return; - } - tool_call.arguments.push_str(&new_args); - tool_call.id.clone() - }; - self.parts.push(ToolCallStreamPart::ToolInputDelta { - id, - delta: new_args, - }); - } - - fn finish_tool_call_at(&mut self, index: usize) { - // Scope the mutable borrow of the tracked tool call, extracting the - // owned data we need to emit the final events. - let (id, function_name, arguments, metadata) = { - let Some(Some(tool_call)) = self.tool_calls.get_mut(index) else { - return; - }; - if tool_call.has_finished { - return; - } - ( - tool_call.id.clone(), - tool_call.function_name.clone(), - tool_call.arguments.clone(), - tool_call.metadata.take(), - ) - }; - if let Some(Some(tool_call)) = self.tool_calls.get_mut(index) { - tool_call.has_finished = true; - } - - self.parts - .push(ToolCallStreamPart::ToolInputEnd { id: id.clone() }); - - let provider_metadata = self - .build_provider_metadata - .as_ref() - .and_then(|build| build(metadata.as_ref())); - - self.parts.push(ToolCallStreamPart::ToolCall { - tool_call_id: id, - tool_name: function_name, - input: arguments, - provider_metadata, - }); - } -} diff --git a/aimux-stream/tests/streaming_tool_call_tracker_test.rs b/aimux-stream/tests/streaming_tool_call_tracker_test.rs deleted file mode 100644 index 25a9b765..00000000 --- a/aimux-stream/tests/streaming_tool_call_tracker_test.rs +++ /dev/null @@ -1,787 +0,0 @@ -//! Rust translation of -//! `packages/provider-utils/src/streaming-tool-call-tracker.test.ts` (18 cases). -//! -//! The TS test collects emitted parts on a shared `controller.enqueue` sink and -//! resets it with `parts.length = 0` between assertions. The Rust equivalent -//! accumulates parts inside the tracker: inspect with `parts()` and reset with -//! `clear_parts()`. - -use aimux_stream::{ - StreamingToolCallDelta, StreamingToolCallTracker, ToolCallStreamPart, TrackerError, - TypeValidation, -}; -use serde_json::{Value, json}; - -/// Convenience builder alias. -fn d() -> StreamingToolCallDelta { - StreamingToolCallDelta::default() -} - -/// A tracker with no metadata handling (`M = ()`), matching the TS tests that -/// don't exercise the metadata hooks. Pinning `M = ()` here lets type -/// inference resolve `ToolCallStreamPart::ToolCall { provider_metadata: None }` -/// in the assertions. -fn new_tracker() -> StreamingToolCallTracker<()> { - StreamingToolCallTracker::<()>::new() -} - -fn tool_call_id(part: &ToolCallStreamPart) -> &str { - match part { - ToolCallStreamPart::ToolCall { tool_call_id, .. } => tool_call_id, - _ => panic!("expected tool-call part"), - } -} - -// == processDelta ========================================================== - -#[test] -fn single_tool_call_accumulated_across_multiple_deltas() { - // TS: "should handle a single tool call accumulated across multiple deltas" - let mut tracker = new_tracker(); - - // First delta: new tool call with id and name. - tracker - .process_delta( - &d().index(0) - .id("call_1") - .tool_type("function") - .function_name("get_weather") - .arguments("{\"ci"), - ) - .unwrap(); - - assert_eq!( - tracker.parts(), - &[ - ToolCallStreamPart::ToolInputStart { - id: "call_1".into(), - tool_name: "get_weather".into(), - }, - ToolCallStreamPart::ToolInputDelta { - id: "call_1".into(), - delta: "{\"ci".into(), - }, - ] - ); - - tracker.clear_parts(); - - // Second delta: more arguments. - tracker - .process_delta(&d().index(0).arguments("ty\": \"San")) - .unwrap(); - - assert_eq!( - tracker.parts(), - &[ToolCallStreamPart::ToolInputDelta { - id: "call_1".into(), - delta: "ty\": \"San".into(), - }] - ); - - tracker.clear_parts(); - - // Third delta: completes the JSON -- must not finalize before flush, since a - // parsable buffer can still be the prefix of longer arguments. - tracker - .process_delta(&d().index(0).arguments(" Francisco\"}")) - .unwrap(); - - assert_eq!( - tracker.parts(), - &[ToolCallStreamPart::ToolInputDelta { - id: "call_1".into(), - delta: " Francisco\"}".into(), - }] - ); - - tracker.clear_parts(); - - tracker.flush(); - - assert_eq!( - tracker.parts(), - &[ - ToolCallStreamPart::ToolInputEnd { - id: "call_1".into(), - }, - ToolCallStreamPart::ToolCall { - tool_call_id: "call_1".into(), - tool_name: "get_weather".into(), - input: "{\"city\": \"San Francisco\"}".into(), - provider_metadata: None, - }, - ] - ); -} - -#[test] -fn full_tool_call_in_single_chunk() { - // TS: "should handle a full tool call in a single chunk" - let mut tracker = new_tracker(); - - tracker - .process_delta( - &d().index(0) - .id("call_1") - .tool_type("function") - .function_name("get_weather") - .arguments("{\"city\": \"London\"}"), - ) - .unwrap(); - - assert_eq!( - tracker.parts(), - &[ - ToolCallStreamPart::ToolInputStart { - id: "call_1".into(), - tool_name: "get_weather".into(), - }, - ToolCallStreamPart::ToolInputDelta { - id: "call_1".into(), - delta: "{\"city\": \"London\"}".into(), - }, - ] - ); - - tracker.clear_parts(); - tracker.flush(); - - assert_eq!( - tracker.parts(), - &[ - ToolCallStreamPart::ToolInputEnd { - id: "call_1".into(), - }, - ToolCallStreamPart::ToolCall { - tool_call_id: "call_1".into(), - tool_name: "get_weather".into(), - input: "{\"city\": \"London\"}".into(), - provider_metadata: None, - }, - ] - ); -} - -#[test] -fn not_finalize_when_argument_prefix_is_parsable_json() { - // TS: "should not finalize a tool call when its argument prefix is parsable JSON" - let mut tracker = new_tracker(); - - tracker - .process_delta( - &d().index(0) - .id("call_1") - .tool_type("function") - .function_name("search") - .arguments("{\"query\": \"test\"}"), - ) - .unwrap(); - - // The parsable prefix must not emit tool-input-end / tool-call. - let types: Vec<&str> = tracker - .parts() - .iter() - .map(|p| match p { - ToolCallStreamPart::ToolInputStart { .. } => "tool-input-start", - ToolCallStreamPart::ToolInputDelta { .. } => "tool-input-delta", - ToolCallStreamPart::ToolInputEnd { .. } => "tool-input-end", - ToolCallStreamPart::ToolCall { .. } => "tool-call", - }) - .collect(); - assert_eq!(types, vec!["tool-input-start", "tool-input-delta"]); - - tracker - .process_delta(&d().index(0).arguments(", \"limit\": 10}")) - .unwrap(); - - tracker.flush(); - - assert_eq!( - tracker.parts().last().unwrap(), - &ToolCallStreamPart::ToolCall { - tool_call_id: "call_1".into(), - tool_name: "search".into(), - input: "{\"query\": \"test\"}, \"limit\": 10}".into(), - provider_metadata: None, - } - ); - - let tool_call_count = tracker - .parts() - .iter() - .filter(|p| matches!(p, ToolCallStreamPart::ToolCall { .. })) - .count(); - assert_eq!(tool_call_count, 1); -} - -#[test] -fn multiple_concurrent_tool_calls() { - // TS: "should handle multiple concurrent tool calls" - let mut tracker = new_tracker(); - - tracker - .process_delta( - &d().index(0) - .id("call_1") - .tool_type("function") - .function_name("get_weather") - .arguments(""), - ) - .unwrap(); - tracker - .process_delta( - &d().index(1) - .id("call_2") - .tool_type("function") - .function_name("get_time") - .arguments(""), - ) - .unwrap(); - - assert_eq!( - tracker.parts(), - &[ - ToolCallStreamPart::ToolInputStart { - id: "call_1".into(), - tool_name: "get_weather".into(), - }, - ToolCallStreamPart::ToolInputStart { - id: "call_2".into(), - tool_name: "get_time".into(), - }, - ] - ); -} - -#[test] -fn skip_deltas_for_already_finished_tool_calls() { - // TS: "should skip deltas for already-finished tool calls" - let mut tracker = new_tracker(); - - tracker - .process_delta( - &d().index(0) - .id("call_1") - .tool_type("function") - .function_name("fn") - .arguments("{}"), - ) - .unwrap(); - - // Finalize via flush. - tracker.flush(); - tracker.clear_parts(); - - // Late delta for the same tool call. - tracker - .process_delta(&d().index(0).arguments("extra")) - .unwrap(); - - assert!(tracker.parts().is_empty()); -} - -#[test] -fn skip_delta_emission_when_arguments_null() { - // TS: "should skip delta emission when arguments are null". - // `arguments: null` maps to `None` (omitted on the builder); behaviorally - // identical to a `function` with no arguments field. - let mut tracker = new_tracker(); - - tracker - .process_delta( - &d().index(0) - .id("call_1") - .tool_type("function") - .function_name("fn") - .arguments(""), - ) - .unwrap(); - - tracker.clear_parts(); - - // Delta with null arguments (no `.arguments(...)` call -> None). - tracker.process_delta(&d().index(0)).unwrap(); - - assert!(tracker.parts().is_empty()); -} - -#[test] -fn use_index_fallback_when_index_not_provided() { - // TS: "should use index fallback when index is not provided" - let mut tracker = new_tracker(); - - tracker - .process_delta( - &d().id("call_1") - .tool_type("function") - .function_name("fn1") - .arguments("{}"), - ) - .unwrap(); - tracker - .process_delta( - &d().id("call_2") - .tool_type("function") - .function_name("fn2") - .arguments("{}"), - ) - .unwrap(); - - let starts: Vec<_> = tracker - .parts() - .iter() - .filter(|p| matches!(p, ToolCallStreamPart::ToolInputStart { .. })) - .cloned() - .collect(); - assert_eq!( - starts, - vec![ - ToolCallStreamPart::ToolInputStart { - id: "call_1".into(), - tool_name: "fn1".into(), - }, - ToolCallStreamPart::ToolInputStart { - id: "call_2".into(), - tool_name: "fn2".into(), - }, - ] - ); -} - -#[test] -fn generate_id_when_id_missing() { - // TS: id is missing → `toolCall.id ?? generateId()` generates a fallback id - // instead of throwing. The default generator returns "generated". - let mut tracker = new_tracker(); - tracker - .process_delta(&d().index(0).tool_type("function").function_name("fn")) - .unwrap(); - - assert_eq!( - tracker.parts(), - &[ToolCallStreamPart::ToolInputStart { - id: "generated".into(), - tool_name: "fn".into(), - }] - ); -} - -#[test] -fn throw_when_function_name_missing() { - // TS: "should throw when function.name is missing". - // `function: {}` (TS) maps to `function: None` here; both yield a missing - // name and the same error. - let mut tracker = new_tracker(); - let err = tracker - .process_delta(&d().index(0).id("call_1").tool_type("function")) - .unwrap_err(); - assert_eq!(err, TrackerError::MissingFunctionName); - assert_eq!(err.to_string(), "Expected 'function.name' to be a string."); -} - -// == typeValidation ======================================================== - -#[test] -fn no_validate_type_with_type_validation_none() { - // TS: "should not validate type with typeValidation: none" - let mut tracker = new_tracker().with_type_validation(TypeValidation::None); - // Should not throw even with a non-function type. - let result = tracker.process_delta( - &d().index(0) - .id("call_1") - .tool_type("custom") - .function_name("fn") - .arguments(""), - ); - assert!(result.is_ok()); -} - -#[test] -fn validate_type_when_present_with_type_validation_if_present() { - // TS: "should validate type when present with typeValidation: if-present" - let mut tracker = new_tracker().with_type_validation(TypeValidation::IfPresent); - - // Should throw for a non-function type. - let err = tracker - .process_delta( - &d().index(0) - .id("call_1") - .tool_type("custom") - .function_name("fn") - .arguments(""), - ) - .unwrap_err(); - assert_eq!(err, TrackerError::InvalidType); - assert_eq!(err.to_string(), "Expected 'function' type."); - - // Should not throw when type is null (absent). The first delta errored - // before creating any tool call, so index 0 is still new. - let result = - tracker.process_delta(&d().index(0).id("call_1").function_name("fn").arguments("")); - assert!(result.is_ok()); -} - -#[test] -fn require_function_type_with_type_validation_required() { - // TS: "should require function type with typeValidation: required" - let mut tracker = new_tracker().with_type_validation(TypeValidation::Required); - - // Should throw when type is null/undefined. - let err = tracker - .process_delta(&d().index(0).id("call_1").function_name("fn").arguments("")) - .unwrap_err(); - assert_eq!(err, TrackerError::InvalidType); - - // Should not throw for 'function' type. - let result = tracker.process_delta( - &d().index(0) - .id("call_1") - .tool_type("function") - .function_name("fn") - .arguments(""), - ); - assert!(result.is_ok()); -} - -// == flush ================================================================= - -#[test] -fn finalize_unfinished_tool_calls_on_flush() { - // TS: "should finalize unfinished tool calls on flush" - let mut tracker = new_tracker(); - - // Start a tool call but don't complete it. - tracker - .process_delta( - &d().index(0) - .id("call_1") - .tool_type("function") - .function_name("fn") - .arguments("{\"key\": \"val"), - ) - .unwrap(); - - tracker.clear_parts(); - - // Flush should finalize. - tracker.flush(); - - assert_eq!( - tracker.parts(), - &[ - ToolCallStreamPart::ToolInputEnd { - id: "call_1".into(), - }, - ToolCallStreamPart::ToolCall { - tool_call_id: "call_1".into(), - tool_name: "fn".into(), - input: "{\"key\": \"val".into(), - provider_metadata: None, - }, - ] - ); -} - -#[test] -fn not_refinalize_already_finished_tool_calls() { - // TS: "should not re-finalize already finished tool calls" - let mut tracker = new_tracker(); - - tracker - .process_delta( - &d().index(0) - .id("call_1") - .tool_type("function") - .function_name("fn") - .arguments("{}"), - ) - .unwrap(); - - // First flush finalizes the tool call. - tracker.flush(); - tracker.clear_parts(); - - tracker.flush(); - - // No events should be emitted since the tool call was already finished. - assert!(tracker.parts().is_empty()); -} - -// == metadata ============================================================= - -#[test] -fn extract_and_include_provider_metadata_in_tool_call_events() { - // TS: "should extract and include provider metadata in tool-call events" - let mut tracker = StreamingToolCallTracker::::new() - .with_extract_metadata(|delta| { - delta - .extra - .get("google")? - .get("thought_signature")? - .as_str() - .map(|s| json!({ "thoughtSignature": s })) - }) - .with_build_provider_metadata(|metadata| { - metadata - .as_ref() - .and_then(|m| m.get("thoughtSignature")) - .and_then(|s| s.as_str()) - .map(|s| json!({ "google": { "thoughtSignature": s } })) - }); - - tracker - .process_delta( - &d().index(0) - .id("call_1") - .tool_type("function") - .function_name("fn") - .arguments("{}") - .extra(json!({ "google": { "thought_signature": "sig123" } })), - ) - .unwrap(); - - tracker.flush(); - - let tool_call = tracker - .parts() - .iter() - .find(|p| matches!(p, ToolCallStreamPart::ToolCall { .. })) - .unwrap(); - assert_eq!( - tool_call, - &ToolCallStreamPart::ToolCall { - tool_call_id: "call_1".into(), - tool_name: "fn".into(), - input: "{}".into(), - provider_metadata: Some(json!({ "google": { "thoughtSignature": "sig123" } })), - } - ); -} - -#[test] -fn include_provider_metadata_for_unfinished_tool_calls_finalized_in_flush() { - // TS: "should include provider metadata for unfinished tool calls finalized in flush" - let mut tracker = StreamingToolCallTracker::::new() - .with_extract_metadata(|_| Some(json!({ "custom": { "key": "value" } }))) - .with_build_provider_metadata(|metadata| { - metadata - .as_ref() - .map(|m| json!({ "provider": (*m).clone() })) - }); - - tracker - .process_delta( - &d().index(0) - .id("call_1") - .tool_type("function") - .function_name("fn") - .arguments("{\"incomplete"), - ) - .unwrap(); - - tracker.clear_parts(); - - tracker.flush(); - - let tool_call = tracker - .parts() - .iter() - .find(|p| matches!(p, ToolCallStreamPart::ToolCall { .. })) - .unwrap(); - assert_eq!( - tool_call, - &ToolCallStreamPart::ToolCall { - tool_call_id: "call_1".into(), - tool_name: "fn".into(), - input: "{\"incomplete".into(), - provider_metadata: Some(json!({ "provider": { "custom": { "key": "value" } } })), - } - ); -} - -#[test] -fn not_include_provider_metadata_when_build_returns_none() { - // TS: "should not include providerMetadata when buildToolCallProviderMetadata returns undefined" - let mut tracker = StreamingToolCallTracker::::new() - .with_extract_metadata(|_| None) - .with_build_provider_metadata(|_| None); - - tracker - .process_delta( - &d().index(0) - .id("call_1") - .tool_type("function") - .function_name("fn") - .arguments("{}"), - ) - .unwrap(); - - tracker.flush(); - - let tool_call = tracker - .parts() - .iter() - .find(|p| matches!(p, ToolCallStreamPart::ToolCall { .. })) - .unwrap(); - assert_eq!( - tool_call, - &ToolCallStreamPart::ToolCall { - tool_call_id: "call_1".into(), - tool_name: "fn".into(), - input: "{}".into(), - provider_metadata: None, - } - ); - // Mirrors the TS `not.toHaveProperty('providerMetadata')` assertion: the - // metadata is explicitly `None`, never `Some`. - assert!(matches!( - tool_call, - ToolCallStreamPart::ToolCall { - provider_metadata: None, - .. - } - )); -} - -// == generateId =========================================================== - -#[test] -fn use_custom_generate_id_for_tool_call_ids_when_id_missing_in_fallback() { - // TS: "should use custom generateId for tool call IDs when id is missing in fallback". - // The id is present (`call_1`), so the custom generator is NOT used; the - // original id is kept (mirrors the TS `toolCall.id ?? generateId()` path). - use std::sync::atomic::{AtomicUsize, Ordering}; - static CALLS: AtomicUsize = AtomicUsize::new(0); - - let mut tracker = StreamingToolCallTracker::<()>::new().with_generate_id(|| { - CALLS.fetch_add(1, Ordering::SeqCst); - "custom-id".to_string() - }); - - tracker - .process_delta( - &d().index(0) - .id("call_1") - .tool_type("function") - .function_name("fn") - .arguments("{\"key\": \"val"), - ) - .unwrap(); - - tracker.clear_parts(); - tracker.flush(); - - let tool_call = tracker - .parts() - .iter() - .find(|p| matches!(p, ToolCallStreamPart::ToolCall { .. })) - .unwrap(); - assert_eq!(tool_call_id(tool_call), "call_1"); - // The custom generator must not have been invoked. - assert_eq!(CALLS.load(Ordering::SeqCst), 0); -} - -#[test] -fn use_custom_generate_id_when_id_is_missing() { - // P2-05: when a tool-call delta omits its id, the configured `generateId` - // fallback is invoked (TS `toolCall.id ?? generateId()`). The generated id - // is used for the emitted ToolInputStart and the finalized ToolCall. - use std::sync::atomic::{AtomicUsize, Ordering}; - static CALLS: AtomicUsize = AtomicUsize::new(0); - - let mut tracker = StreamingToolCallTracker::<()>::new().with_generate_id(|| { - CALLS.fetch_add(1, Ordering::SeqCst); - "custom-id".to_string() - }); - - // No `.id(...)` — the generator must fill it in. - tracker - .process_delta( - &d().index(0) - .tool_type("function") - .function_name("fn") - .arguments("{\"key\": \"val"), - ) - .unwrap(); - - // The generator was invoked exactly once for the missing id. - assert_eq!(CALLS.load(Ordering::SeqCst), 1); - - // The emitted start part carries the generated id. - assert_eq!( - tracker.parts(), - &[ - ToolCallStreamPart::ToolInputStart { - id: "custom-id".into(), - tool_name: "fn".into(), - }, - ToolCallStreamPart::ToolInputDelta { - id: "custom-id".into(), - delta: "{\"key\": \"val".into(), - }, - ] - ); - - tracker.clear_parts(); - tracker.flush(); - - let tool_call = tracker - .parts() - .iter() - .find(|p| matches!(p, ToolCallStreamPart::ToolCall { .. })) - .unwrap(); - assert_eq!(tool_call_id(tool_call), "custom-id"); - // No additional generator invocations during flush. - assert_eq!(CALLS.load(Ordering::SeqCst), 1); -} - -// == index bound (P1-10) ================================================== - -#[test] -fn reject_tool_call_index_above_max_index() { - // A remote `index` far above the cap must not resize `tool_calls` to a - // huge vector; it returns `IndexOutOfRange`. - let mut tracker = new_tracker().with_max_index(4); - - // index == max_index is accepted (boundary). - tracker - .process_delta( - &d().index(4) - .id("call_4") - .tool_type("function") - .function_name("fn") - .arguments(""), - ) - .unwrap(); - - // index == max_index + 1 is rejected. - let err = tracker - .process_delta( - &d().index(5) - .id("call_5") - .tool_type("function") - .function_name("fn") - .arguments(""), - ) - .unwrap_err(); - assert_eq!(err, TrackerError::IndexOutOfRange); - assert_eq!(err.to_string(), "Tool call index out of range"); -} - -#[test] -fn default_max_index_rejects_huge_index() { - // The default cap (1024) rejects an absurd index that would otherwise - // resize `tool_calls` to index+1 slots. - let mut tracker = new_tracker(); - let err = tracker - .process_delta( - &d().index(1_000_000) - .id("call_x") - .tool_type("function") - .function_name("fn") - .arguments(""), - ) - .unwrap_err(); - assert_eq!(err, TrackerError::IndexOutOfRange); -} From d72cfc3bf083f4e11558288f854b052503959308 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 05:02:49 +0000 Subject: [PATCH 2/9] feat(stream)!: align StreamingToolCallTracker with the current AI SDK Re-adds the tracker as a port of the current @ai-sdk/provider-utils streaming-tool-call-tracker.ts instead of the older index-only version removed in the previous commit. - Correlate deltas by wire id, index and function name; drop ambiguous ones - Unique tool-call ids with bounded suffixes; blank generator output falls back to "tool-call" - Ignore unmatched blank function names; retain blank-name continuations - Flush in index order only when every call has an index - New StreamingToolCallArgumentState / starts_with_structured_value port - process_delta and flush return the emitted parts instead of buffering Tests are ported case for case from the upstream suites (45 tracker cases, 9 argument-state cases). No caller in the workspace uses the tracker yet. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_015EYDYDWYcPsDmjBjFeuDVe --- CHANGELOG.md | 17 +- aimux-stream/src/lib.rs | 9 + .../src/streaming_tool_call_argument_state.rs | 116 ++ .../src/streaming_tool_call_tracker.rs | 638 ++++++++++ ...streaming_tool_call_argument_state_test.rs | 73 ++ .../tests/streaming_tool_call_tracker_test.rs | 1098 +++++++++++++++++ 6 files changed, 1947 insertions(+), 4 deletions(-) create mode 100644 aimux-stream/src/streaming_tool_call_argument_state.rs create mode 100644 aimux-stream/src/streaming_tool_call_tracker.rs create mode 100644 aimux-stream/tests/streaming_tool_call_argument_state_test.rs create mode 100644 aimux-stream/tests/streaming_tool_call_tracker_test.rs diff --git a/CHANGELOG.md b/CHANGELOG.md index a8b303ad..288c76a2 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -11,10 +11,19 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 **Rust (aimux-stream)** -- Removed `StreamingToolCallTracker` and its companion types - (`StreamingToolCallDelta`, `StreamingToolCallFunction`, `ToolCallStreamPart`, - `TrackerError`, `TypeValidation`). No crate in the workspace used them: - providers accumulate streamed tool-call deltas themselves. +- `StreamingToolCallTracker` rewritten to match the current AI SDK + (`@ai-sdk/provider-utils`) design; the previous version was a port of an + older, index-only tracker that nothing in the workspace used. Deltas are now + correlated by wire `id`, `index` and function name (ambiguous deltas are + dropped), ids are de-duplicated with bounded suffixes, blank function names + are ignored, and `flush` orders calls by index only when every call has one. + API changes: `process_delta` and `flush` now return the emitted + `ToolCallStreamPart`s instead of buffering them (`parts()` / `clear_parts()` + are gone); `with_max_index` and `TrackerError::{MissingId, IndexOutOfRange}` + are removed; `TrackerError::IdExhausted` is added; the builder closures must + be `Send + Sync`. +- Added `StreamingToolCallArgumentState` and `starts_with_structured_value`, + the structural JSON-prefix tracker the new correlation logic relies on. ## [0.5.0] - 2026-09-27 diff --git a/aimux-stream/src/lib.rs b/aimux-stream/src/lib.rs index 19d442c5..d506e0ad 100644 --- a/aimux-stream/src/lib.rs +++ b/aimux-stream/src/lib.rs @@ -6,8 +6,17 @@ pub mod lines; pub mod ndjson; pub mod sse; +pub mod streaming_tool_call_argument_state; +pub mod streaming_tool_call_tracker; // Re-export the most commonly used items. pub use lines::extract_lines; pub use ndjson::{NdjsonError, NdjsonStream}; pub use sse::{SseError, SseEvent, SseStream}; +pub use streaming_tool_call_argument_state::{ + StreamingToolCallArgumentState, starts_with_structured_value, +}; +pub use streaming_tool_call_tracker::{ + StreamingToolCallDelta, StreamingToolCallFunction, StreamingToolCallTracker, + ToolCallStreamPart, TrackerError, TypeValidation, +}; diff --git a/aimux-stream/src/streaming_tool_call_argument_state.rs b/aimux-stream/src/streaming_tool_call_argument_state.rs new file mode 100644 index 00000000..db156e66 --- /dev/null +++ b/aimux-stream/src/streaming_tool_call_argument_state.rs @@ -0,0 +1,116 @@ +//! Incremental structure tracking for streamed tool-call arguments. +//! +//! Rust translation of `@ai-sdk/provider-utils`'s +//! `StreamingToolCallArgumentState` +//! (`packages/provider-utils/src/streaming-tool-call-argument-state.ts`). +//! +//! The state is intentionally *structural* rather than a JSON parse: a +//! currently parsable scalar can still be the prefix of a later value, but a +//! closed top-level `{…}` / `[…]` cannot be extended. + +#[derive(Debug, Clone, PartialEq, Eq)] +enum ArgumentStructure { + Undetermined, + Other, + Structured { + stack: Vec, + in_string: bool, + escaped: bool, + complete: bool, + }, +} + +/// Whether `value` (ignoring leading whitespace) starts with `{` or `[`. +#[must_use] +pub fn starts_with_structured_value(value: Option<&str>) -> bool { + value + .and_then(|v| v.trim_start().chars().next()) + .is_some_and(|c| c == '{' || c == '[') +} + +/// Incrementally tracks whether streamed tool-call arguments contain a +/// complete structured JSON value. +#[derive(Debug, Clone)] +pub struct StreamingToolCallArgumentState { + structure: ArgumentStructure, +} + +impl StreamingToolCallArgumentState { + /// Create a state seeded with `initial_value`. + #[must_use] + pub fn new(initial_value: &str) -> Self { + let mut state = Self { + structure: ArgumentStructure::Undetermined, + }; + state.append(initial_value); + state + } + + /// `true` once a top-level `{…}` / `[…]` value has been closed. + #[must_use] + pub fn has_complete_structured_value(&self) -> bool { + matches!( + self.structure, + ArgumentStructure::Structured { complete: true, .. } + ) + } + + /// Feed the next argument fragment. + pub fn append(&mut self, delta: &str) { + for character in delta.chars() { + match &mut self.structure { + ArgumentStructure::Undetermined => { + if character.is_whitespace() { + continue; + } + self.structure = if character == '{' || character == '[' { + ArgumentStructure::Structured { + stack: vec![character], + in_string: false, + escaped: false, + complete: false, + } + } else { + ArgumentStructure::Other + }; + } + ArgumentStructure::Other | ArgumentStructure::Structured { complete: true, .. } => { + } + ArgumentStructure::Structured { + stack, + in_string, + escaped, + complete, + } => { + if *in_string { + if *escaped { + *escaped = false; + } else if character == '\\' { + *escaped = true; + } else if character == '"' { + *in_string = false; + } + continue; + } + + match character { + '"' => *in_string = true, + '{' | '[' => stack.push(character), + '}' | ']' => { + let expected = if character == '}' { '{' } else { '[' }; + if stack.last() != Some(&expected) { + self.structure = ArgumentStructure::Other; + continue; + } + stack.pop(); + if stack.is_empty() { + *complete = true; + } + } + _ => {} + } + } + } + } + } +} diff --git a/aimux-stream/src/streaming_tool_call_tracker.rs b/aimux-stream/src/streaming_tool_call_tracker.rs new file mode 100644 index 00000000..9a8d7240 --- /dev/null +++ b/aimux-stream/src/streaming_tool_call_tracker.rs @@ -0,0 +1,638 @@ +//! Streaming tool call tracker. +//! +//! Rust translation of `@ai-sdk/provider-utils`'s `StreamingToolCallTracker` +//! (`packages/provider-utils/src/streaming-tool-call-tracker.ts`). +//! +//! Tracks streaming tool call state across the deltas of an OpenAI-compatible +//! chat completion stream: accumulates `arguments` fragments, emits +//! `tool-input-start` / `tool-input-delta` / `tool-input-end` / `tool-call` +//! parts, and finalizes unfinished calls on [`StreamingToolCallTracker::flush`]. +//! +//! Deltas are correlated to calls by wire `id`, `index` and function name +//! (see [`StreamingToolCallTracker`]'s resolution table), not by `index` +//! alone, so providers that reuse indices, repeat ids, drop ids on +//! continuations or send blank names are handled. +//! +//! Like the TS original, a call is *never* finalized before `flush`: a +//! parsable argument buffer can still be the prefix of a longer argument +//! string, so acting on it early would use truncated inputs (ai-sdk #13137). + +use std::collections::{HashMap, HashSet}; + +use serde_json::Value; +use thiserror::Error; + +use crate::streaming_tool_call_argument_state::{ + StreamingToolCallArgumentState, starts_with_structured_value, +}; + +/// Fallback id used when the id generator returns a blank string. +const FALLBACK_TOOL_CALL_ID: &str = "tool-call"; + +/// The `function` sub-object of a streaming tool call delta. +#[derive(Debug, Clone, Default)] +pub struct StreamingToolCallFunction { + pub name: Option, + pub arguments: Option, +} + +/// A streaming tool call delta — the `tool_calls[i]` entry of an OpenAI-style +/// streaming chunk. +/// +/// `arguments: null` (TS) maps to `None`; `arguments: ''` maps to `Some("")`. +#[derive(Debug, Clone, Default)] +pub struct StreamingToolCallDelta { + pub index: Option, + pub id: Option, + /// The `type` field. Named `r#type` because `type` is a reserved word. + pub r#type: Option, + pub function: Option, + /// Provider-specific payload carried alongside the standard fields, read + /// by the `extract_metadata` hook (e.g. a Google thought signature). + pub extra: Value, +} + +impl StreamingToolCallDelta { + #[must_use] + pub fn new() -> Self { + Self::default() + } + + #[must_use] + pub fn index(mut self, index: usize) -> Self { + self.index = Some(index); + self + } + + #[must_use] + pub fn id(mut self, id: impl Into) -> Self { + self.id = Some(id.into()); + self + } + + /// Set the `type` field (named `tool_type` because `type` is reserved). + #[must_use] + pub fn tool_type(mut self, t: impl Into) -> Self { + self.r#type = Some(t.into()); + self + } + + #[must_use] + pub fn function_name(mut self, name: impl Into) -> Self { + self.function.get_or_insert_with(Default::default).name = Some(name.into()); + self + } + + /// Set the `function.arguments` fragment. Pass `""` for an explicit empty + /// fragment; omit the call for `None` (TS `arguments: null`). + #[must_use] + pub fn arguments(mut self, args: impl Into) -> Self { + self.function.get_or_insert_with(Default::default).arguments = Some(args.into()); + self + } + + #[must_use] + pub fn extra(mut self, extra: Value) -> Self { + self.extra = extra; + self + } +} + +/// The stream parts emitted by [`StreamingToolCallTracker`]. +/// +/// Mirrors the subset of `LanguageModelV4StreamPart` the TS tracker enqueues. +/// [`ToolCallStreamPart::ToolCall::input`] is the raw accumulated argument +/// *string*; the tracker does not parse it. +#[derive(Debug, Clone, PartialEq)] +pub enum ToolCallStreamPart { + /// Start of a tool call's input streaming. + ToolInputStart { id: String, tool_name: String }, + /// A partial argument fragment. + ToolInputDelta { id: String, delta: String }, + /// End of a tool call's input streaming. + ToolInputEnd { id: String }, + /// A complete, finalized tool call. + ToolCall { + tool_call_id: String, + tool_name: String, + input: String, + provider_metadata: Option, + }, +} + +/// How to validate the `type` field on a new tool call delta. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub enum TypeValidation { + /// No validation (default). + #[default] + None, + /// Error if `type` is present and not `"function"`. + IfPresent, + /// Error if `type` is not exactly `"function"`. + Required, +} + +/// Errors raised while processing a tool call delta. +#[derive(Debug, Error, PartialEq, Eq)] +pub enum TrackerError { + #[error("Expected 'function.name' to be a string.")] + MissingFunctionName, + #[error("Expected 'function' type.")] + InvalidType, + #[error("Failed to create a unique tool call ID.")] + IdExhausted, +} + +struct TrackedToolCall { + id: String, + index: Option, + sequence: usize, + function_name: String, + arguments: String, + argument_state: StreamingToolCallArgumentState, + has_finished: bool, + metadata: Option, +} + +enum ToolCallResolution { + Existing(usize), + New, + Ambiguous, +} + +type GenerateIdFn = Box String + Send + Sync>; +type ExtractMetadataFn = Box Option + Send + Sync>; +type BuildMetadataFn = Box) -> Option + Send + Sync>; + +/// Tracks streaming tool call state across multiple deltas. +/// +/// [`process_delta`](Self::process_delta) and [`flush`](Self::flush) return +/// the parts to forward downstream (the TS tracker enqueues them on a +/// controller instead). +/// +/// The type parameter `M` is the provider-metadata type; use `()` (the +/// default) when no metadata handling is needed. +/// +/// # Correlation +/// +/// | ID evidence | index/name evidence | start evidence | resolution | +/// | --- | --- | --- | --- | +/// | known | matching | any | matching call, new call, or ambiguity | +/// | known | conflicting | named | new call | +/// | unseen | matching | structured start | new call | +/// | unseen | matching | continuation | matching call or ambiguity | +/// | absent | matching | any | matching call, new call, or ambiguity | +/// | absent | absent | named | new call | +/// | absent | absent | unnamed | sole unfinished call, new call, or ambiguity | +/// +/// An *ambiguous* delta is dropped. +pub struct StreamingToolCallTracker { + tool_calls: Vec>, + tool_calls_by_id: HashMap>, + tool_calls_by_index: HashMap>, + used_tool_call_ids: HashSet, + next_generated_id_suffixes: HashMap, + generate_id: GenerateIdFn, + type_validation: TypeValidation, + extract_metadata: Option>, + build_provider_metadata: Option>, +} + +impl Default for StreamingToolCallTracker { + fn default() -> Self { + Self::new() + } +} + +impl StreamingToolCallTracker { + /// Create a tracker with no metadata handling and default settings. + /// + /// The default id generator returns a blank-free constant, so ids that + /// need generating become `tool-call`, `tool-call-1`, `tool-call-2`, … + #[must_use] + pub fn new() -> Self { + Self { + tool_calls: Vec::new(), + tool_calls_by_id: HashMap::new(), + tool_calls_by_index: HashMap::new(), + used_tool_call_ids: HashSet::new(), + next_generated_id_suffixes: HashMap::new(), + generate_id: Box::new(|| FALLBACK_TOOL_CALL_ID.to_string()), + type_validation: TypeValidation::None, + extract_metadata: None, + build_provider_metadata: None, + } + } + + /// Set the id generator (the TS `generateId` option). Blank or repeated + /// outputs are turned into usable unique ids. + #[must_use] + pub fn with_generate_id String + Send + Sync + 'static>(mut self, f: F) -> Self { + self.generate_id = Box::new(f); + self + } + + /// Set the `type` validation mode (the TS `typeValidation` option). + #[must_use] + pub fn with_type_validation(mut self, v: TypeValidation) -> Self { + self.type_validation = v; + self + } + + /// Set the metadata extractor (the TS `extractMetadata` option). Called + /// once when a new tool call is detected. + #[must_use] + pub fn with_extract_metadata< + F: Fn(&StreamingToolCallDelta) -> Option + Send + Sync + 'static, + >( + mut self, + f: F, + ) -> Self { + self.extract_metadata = Some(Box::new(f)); + self + } + + /// Set the provider-metadata builder (the TS + /// `buildToolCallProviderMetadata` option). If it returns `None`, the + /// `tool-call` part carries no `provider_metadata`. + #[must_use] + pub fn with_build_provider_metadata) -> Option + Send + Sync + 'static>( + mut self, + f: F, + ) -> Self { + self.build_provider_metadata = Some(Box::new(f)); + self + } + + /// Process a tool call delta from a streaming chunk and return the parts + /// it produces (possibly none). + /// + /// # Errors + /// + /// Returns [`TrackerError::InvalidType`] when `type` validation fails, + /// [`TrackerError::MissingFunctionName`] when a new call has no + /// (non-null) function name, and [`TrackerError::IdExhausted`] if no + /// unique id can be produced. + pub fn process_delta( + &mut self, + delta: &StreamingToolCallDelta, + ) -> Result>, TrackerError> { + let wire_name = delta.function.as_ref().and_then(|f| f.name.as_deref()); + let has_blank_name = wire_name.is_some_and(|n| n.trim().is_empty()); + let wire_id = non_blank(delta.id.as_deref()); + let name = non_blank(wire_name); + let index = delta.index; + let arguments = delta.function.as_ref().and_then(|f| f.arguments.as_deref()); + + let resolution = self.resolve_tool_call( + wire_id, + index, + name, + name.is_some() && starts_with_structured_value(arguments), + ); + + let mut parts = Vec::new(); + let call = match resolution { + ToolCallResolution::Ambiguous => return Ok(parts), + ToolCallResolution::New => { + // Blank names cannot start a usable call, but some providers + // repeat a blank name on continuations. Those were correlated + // above; only an unmatched blank-name delta is ignored. + if has_blank_name { + return Ok(parts); + } + self.process_new_tool_call(delta, wire_id, index, name, &mut parts)? + } + ToolCallResolution::Existing(call) => { + if let Some(wire_id) = wire_id { + self.associate_wire_id(call, wire_id); + } + self.process_existing_tool_call(call, arguments, &mut parts); + call + } + }; + + if let Some(index) = index { + self.tool_calls_by_index + .entry(index) + .or_default() + .insert(call); + } + Ok(parts) + } + + /// Finalize any unfinished tool calls and return the closing parts. Call + /// once when the stream ends. + pub fn flush(&mut self) -> Vec> { + // Index order is only reliable when every call has an index; for + // mixed streams keep insertion order. + let mut order: Vec = (0..self.tool_calls.len()).collect(); + if self.tool_calls.iter().all(|c| c.index.is_some()) { + order.sort_by_key(|&i| (self.tool_calls[i].index, self.tool_calls[i].sequence)); + } + + let mut parts = Vec::new(); + for call in order { + if !self.tool_calls[call].has_finished { + self.finish_tool_call(call, &mut parts); + } + } + parts + } + + fn resolve_tool_call( + &self, + wire_id: Option<&str>, + index: Option, + name: Option<&str>, + has_explicit_call_start: bool, + ) -> ToolCallResolution { + let indexed = index.and_then(|i| self.tool_calls_by_index.get(&i)); + let matching_indexed = self.filter_by_name(indexed, name); + + if let Some(wire_id) = wire_id { + if let Some(with_id) = self.tool_calls_by_id.get(wire_id) { + if index.is_some() { + let matching: Vec = matching_indexed + .iter() + .copied() + .filter(|c| with_id.contains(c)) + .collect(); + let resolved = self.resolve_matching(matching, has_explicit_call_start); + if !matches!(resolved, ToolCallResolution::New) { + return resolved; + } + + // A named delta with a distinct index starts a new call + // even when its wire id and name repeat: providers may + // reuse ids across parallel calls. + if name.is_some() { + return ToolCallResolution::New; + } + + // Conflicting labels on a continuation cannot be + // resolved safely. + if indexed.is_some() { + return ToolCallResolution::Ambiguous; + } + + return self.resolve_matching(sorted(with_id), false); + } + + if let Some(name) = name { + let matching: Vec = sorted(with_id) + .into_iter() + .filter(|&c| self.tool_calls[c].function_name == name) + .collect(); + return self.resolve_matching(matching, has_explicit_call_start); + } + + return self.resolve_matching(sorted(with_id), false); + } + + if !matching_indexed.is_empty() { + // A previously unseen id plus a named structured argument + // start is stronger evidence of a distinct call than a reused + // index/name; ids may still change on plain continuations. + return if has_explicit_call_start { + ToolCallResolution::New + } else { + self.resolve_matching(matching_indexed, false) + }; + } + + return ToolCallResolution::New; + } + + if indexed.is_some() { + // Repeated names are valid on continuations; a different name at + // the same index means a new call from a provider that reuses + // indices across parallel calls. + return self.resolve_matching(matching_indexed, has_explicit_call_start); + } + + if name.is_some() { + return ToolCallResolution::New; + } + + let unfinished: Vec = (0..self.tool_calls.len()) + .filter(|&c| !self.tool_calls[c].has_finished) + .collect(); + match unfinished.len() { + 0 => ToolCallResolution::New, + 1 => ToolCallResolution::Existing(unfinished[0]), + _ => ToolCallResolution::Ambiguous, + } + } + + fn filter_by_name(&self, calls: Option<&HashSet>, name: Option<&str>) -> Vec { + calls.map_or_else(Vec::new, |calls| { + sorted(calls) + .into_iter() + .filter(|&c| name.is_none_or(|n| self.tool_calls[c].function_name == n)) + .collect() + }) + } + + fn resolve_matching( + &self, + calls: Vec, + has_explicit_call_start: bool, + ) -> ToolCallResolution { + if calls.is_empty() { + return ToolCallResolution::New; + } + + if !has_explicit_call_start { + return if calls.len() == 1 { + ToolCallResolution::Existing(calls[0]) + } else { + ToolCallResolution::Ambiguous + }; + } + + // A repeated name can occur on continuations. A fresh structured + // argument prefix signals another call only once the matching call + // has completed its own structured payload. + let continuable: Vec = calls + .into_iter() + .filter(|&c| { + !self.tool_calls[c] + .argument_state + .has_complete_structured_value() + }) + .collect(); + match continuable.len() { + 0 => ToolCallResolution::New, + 1 => ToolCallResolution::Existing(continuable[0]), + _ => ToolCallResolution::Ambiguous, + } + } + + fn process_new_tool_call( + &mut self, + delta: &StreamingToolCallDelta, + wire_id: Option<&str>, + index: Option, + name: Option<&str>, + parts: &mut Vec>, + ) -> Result { + match self.type_validation { + TypeValidation::Required => { + if delta.r#type.as_deref() != Some("function") { + return Err(TrackerError::InvalidType); + } + } + TypeValidation::IfPresent => { + if delta.r#type.as_deref().is_some_and(|t| t != "function") { + return Err(TrackerError::InvalidType); + } + } + TypeValidation::None => {} + } + + let name = name.ok_or(TrackerError::MissingFunctionName)?; + let id = self.create_tool_call_id(wire_id)?; + + parts.push(ToolCallStreamPart::ToolInputStart { + id: id.clone(), + tool_name: name.to_string(), + }); + + let metadata = self + .extract_metadata + .as_ref() + .and_then(|extract| extract(delta)); + + let initial_arguments = delta + .function + .as_ref() + .and_then(|f| f.arguments.clone()) + .unwrap_or_default(); + + let call = self.tool_calls.len(); + self.tool_calls.push(TrackedToolCall { + id: id.clone(), + index, + sequence: call, + function_name: name.to_string(), + argument_state: StreamingToolCallArgumentState::new(&initial_arguments), + arguments: initial_arguments.clone(), + has_finished: false, + metadata, + }); + if let Some(wire_id) = wire_id { + self.associate_wire_id(call, wire_id); + } + + if !initial_arguments.is_empty() { + parts.push(ToolCallStreamPart::ToolInputDelta { + id, + delta: initial_arguments, + }); + } + + // Tool calls must not finalize before the stream ends (ai-sdk + // #13137); finalization happens in `flush`. + Ok(call) + } + + fn process_existing_tool_call( + &mut self, + call: usize, + arguments: Option<&str>, + parts: &mut Vec>, + ) { + let tool_call = &mut self.tool_calls[call]; + if tool_call.has_finished { + return; + } + if let Some(arguments) = arguments { + tool_call.argument_state.append(arguments); + tool_call.arguments.push_str(arguments); + parts.push(ToolCallStreamPart::ToolInputDelta { + id: tool_call.id.clone(), + delta: arguments.to_string(), + }); + } + } + + fn associate_wire_id(&mut self, call: usize, wire_id: &str) { + self.tool_calls_by_id + .entry(wire_id.to_string()) + .or_default() + .insert(call); + } + + fn create_tool_call_id(&mut self, wire_id: Option<&str>) -> Result { + if let Some(wire_id) = wire_id + && !self.used_tool_call_ids.contains(wire_id) + { + self.used_tool_call_ids.insert(wire_id.to_string()); + return Ok(wire_id.to_string()); + } + + let generated = non_blank(Some(&(self.generate_id)())) + .unwrap_or(FALLBACK_TOOL_CALL_ID) + .to_string(); + + if !self.used_tool_call_ids.contains(&generated) { + self.used_tool_call_ids.insert(generated.clone()); + return Ok(generated); + } + + // Resume after the last suffix checked for this generated value so + // deterministic generators stay bounded without rescanning occupied + // suffixes. + let initial_suffix = self + .next_generated_id_suffixes + .get(&generated) + .copied() + .unwrap_or(1); + let maximum_suffix = initial_suffix + self.used_tool_call_ids.len(); + for suffix in initial_suffix..=maximum_suffix { + let suffixed = format!("{generated}-{suffix}"); + if !self.used_tool_call_ids.contains(&suffixed) { + self.used_tool_call_ids.insert(suffixed.clone()); + self.next_generated_id_suffixes + .insert(generated, suffix + 1); + return Ok(suffixed); + } + } + + // Unreachable by the pigeonhole principle; guards the invariant + // instead of looping without bound. + Err(TrackerError::IdExhausted) + } + + fn finish_tool_call(&mut self, call: usize, parts: &mut Vec>) { + let tool_call = &mut self.tool_calls[call]; + tool_call.has_finished = true; + + parts.push(ToolCallStreamPart::ToolInputEnd { + id: tool_call.id.clone(), + }); + + let provider_metadata = self + .build_provider_metadata + .as_ref() + .and_then(|build| build(tool_call.metadata.as_ref())); + + parts.push(ToolCallStreamPart::ToolCall { + tool_call_id: tool_call.id.clone(), + tool_name: tool_call.function_name.clone(), + input: tool_call.arguments.clone(), + provider_metadata, + }); + } +} + +fn non_blank(value: Option<&str>) -> Option<&str> { + value.filter(|v| !v.trim().is_empty()) +} + +fn sorted(calls: &HashSet) -> Vec { + let mut calls: Vec = calls.iter().copied().collect(); + calls.sort_unstable(); + calls +} diff --git a/aimux-stream/tests/streaming_tool_call_argument_state_test.rs b/aimux-stream/tests/streaming_tool_call_argument_state_test.rs new file mode 100644 index 00000000..44cf754e --- /dev/null +++ b/aimux-stream/tests/streaming_tool_call_argument_state_test.rs @@ -0,0 +1,73 @@ +//! Port of `streaming-tool-call-argument-state.test.ts` from +//! `@ai-sdk/provider-utils`. + +use aimux_stream::{StreamingToolCallArgumentState, starts_with_structured_value}; + +#[test] +fn starts_with_structured_value_true() { + for value in ["{}", " {", "[]", "\n["] { + assert!(starts_with_structured_value(Some(value)), "{value:?}"); + } +} + +#[test] +fn starts_with_structured_value_false() { + assert!(!starts_with_structured_value(None)); + for value in ["", " ", "1", "\"value\""] { + assert!(!starts_with_structured_value(Some(value)), "{value:?}"); + } +} + +#[test] +fn tracks_a_structured_value_across_deltas() { + let mut state = StreamingToolCallArgumentState::new(" {\"value\":"); + assert!(!state.has_complete_structured_value()); + + state.append("1}"); + assert!(state.has_complete_structured_value()); +} + +#[test] +fn tracks_nested_objects_and_arrays() { + let state = StreamingToolCallArgumentState::new("[{\"value\":{\"items\":[1,2]}}]"); + assert!(state.has_complete_structured_value()); +} + +#[test] +fn ignores_structural_characters_inside_strings() { + let state = StreamingToolCallArgumentState::new("{\"value\":\"braces: } ] { [\"}"); + assert!(state.has_complete_structured_value()); +} + +#[test] +fn handles_escaped_quotes_across_deltas() { + let mut state = StreamingToolCallArgumentState::new("{\"value\":\"escaped quote: \\\""); + assert!(!state.has_complete_structured_value()); + + state.append(" still in string\"}"); + assert!(state.has_complete_structured_value()); +} + +#[test] +fn can_begin_after_an_empty_or_whitespace_only_delta() { + let mut state = StreamingToolCallArgumentState::new(" "); + + state.append("["); + assert!(!state.has_complete_structured_value()); + + state.append("]"); + assert!(state.has_complete_structured_value()); +} + +#[test] +fn does_not_treat_scalar_arguments_as_a_complete_structured_value() { + let state = StreamingToolCallArgumentState::new("12"); + assert!(!state.has_complete_structured_value()); +} + +#[test] +fn does_not_recover_mismatched_structures_as_complete() { + let mut state = StreamingToolCallArgumentState::new("{\"value\":]"); + state.append("}"); + assert!(!state.has_complete_structured_value()); +} diff --git a/aimux-stream/tests/streaming_tool_call_tracker_test.rs b/aimux-stream/tests/streaming_tool_call_tracker_test.rs new file mode 100644 index 00000000..7c55044c --- /dev/null +++ b/aimux-stream/tests/streaming_tool_call_tracker_test.rs @@ -0,0 +1,1098 @@ +//! Port of `streaming-tool-call-tracker.test.ts` from +//! `@ai-sdk/provider-utils`, case for case. + +use std::sync::Arc; +use std::sync::atomic::{AtomicUsize, Ordering}; + +use aimux_stream::{ + StreamingToolCallDelta, StreamingToolCallFunction, StreamingToolCallTracker, + ToolCallStreamPart, TrackerError, TypeValidation, +}; +use serde_json::{Value, json}; + +type Part = ToolCallStreamPart; + +/// Tracker plus the parts it has emitted so far (the TS `createCollector`). +struct Harness { + tracker: StreamingToolCallTracker, + parts: Vec>, +} + +impl Harness<()> { + fn new() -> Self { + Self::with(StreamingToolCallTracker::new()) + } + + fn with(tracker: StreamingToolCallTracker<()>) -> Self { + Self::with_meta(tracker) + } +} + +impl Harness { + fn with_meta(tracker: StreamingToolCallTracker) -> Self { + Self { + tracker, + parts: Vec::new(), + } + } + + fn delta(&mut self, delta: StreamingToolCallDelta) -> Result<(), TrackerError> { + let parts = self.tracker.process_delta(&delta)?; + self.parts.extend(parts); + Ok(()) + } + + fn flush(&mut self) { + let parts = self.tracker.flush(); + self.parts.extend(parts); + } + + fn clear(&mut self) { + self.parts.clear(); + } +} + +impl Harness { + /// `(id, name, input)` of every emitted `tool-call`, in order. + fn tool_calls(&self) -> Vec<(String, String, String)> { + self.parts + .iter() + .filter_map(|part| match part { + Part::ToolCall { + tool_call_id, + tool_name, + input, + .. + } => Some((tool_call_id.clone(), tool_name.clone(), input.clone())), + _ => None, + }) + .collect() + } +} + +/// A delta with an explicit `function` object (as every TS test passes one). +fn delta( + index: Option, + id: Option<&str>, + ty: Option<&str>, + name: Option<&str>, + arguments: Option<&str>, +) -> StreamingToolCallDelta { + StreamingToolCallDelta { + index, + id: id.map(str::to_string), + r#type: ty.map(str::to_string), + function: Some(StreamingToolCallFunction { + name: name.map(str::to_string), + arguments: arguments.map(str::to_string), + }), + extra: Value::Null, + } +} + +/// Full call start: index, id, `type: 'function'`, name and arguments. +fn start(index: usize, id: &str, name: &str, arguments: &str) -> StreamingToolCallDelta { + delta( + Some(index), + Some(id), + Some("function"), + Some(name), + Some(arguments), + ) +} + +/// Continuation carrying only arguments (plus an optional index). +fn cont(index: Option, arguments: &str) -> StreamingToolCallDelta { + delta(index, None, None, None, Some(arguments)) +} + +fn tc(id: &str, name: &str, input: &str) -> (String, String, String) { + (id.into(), name.into(), input.into()) +} + +fn input_start(id: &str, tool_name: &str) -> Part { + Part::ToolInputStart { + id: id.into(), + tool_name: tool_name.into(), + } +} + +fn input_delta(id: &str, delta: &str) -> Part { + Part::ToolInputDelta { + id: id.into(), + delta: delta.into(), + } +} + +fn input_end(id: &str) -> Part { + Part::ToolInputEnd { id: id.into() } +} + +fn tool_call(id: &str, name: &str, input: &str) -> Part { + Part::ToolCall { + tool_call_id: id.into(), + tool_name: name.into(), + input: input.into(), + provider_metadata: None, + } +} + +/// Deterministic id generator: `prefix-1`, `prefix-2`, … and a call counter. +fn counting_generator(ids: &'static [&'static str]) -> (impl Fn() -> String, Arc) { + let calls = Arc::new(AtomicUsize::new(0)); + let counter = Arc::clone(&calls); + ( + move || { + let n = counter.fetch_add(1, Ordering::SeqCst); + ids[n.min(ids.len() - 1)].to_string() + }, + calls, + ) +} + +mod process_delta { + use super::*; + + #[test] + fn single_tool_call_accumulated_across_multiple_deltas() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "get_weather", "{\"ci")).unwrap(); + assert_eq!( + h.parts, + vec![ + input_start("call_1", "get_weather"), + input_delta("call_1", "{\"ci"), + ] + ); + h.clear(); + + h.delta(cont(Some(0), "ty\": \"San")).unwrap(); + assert_eq!(h.parts, vec![input_delta("call_1", "ty\": \"San")]); + h.clear(); + + // Completing the JSON must not finalize before flush: a parsable + // buffer can still be the prefix of longer arguments. + h.delta(cont(Some(0), " Francisco\"}")).unwrap(); + assert_eq!(h.parts, vec![input_delta("call_1", " Francisco\"}")]); + h.clear(); + + h.flush(); + assert_eq!( + h.parts, + vec![ + input_end("call_1"), + tool_call("call_1", "get_weather", "{\"city\": \"San Francisco\"}"), + ] + ); + } + + #[test] + fn full_tool_call_in_a_single_chunk() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "get_weather", "{\"city\": \"London\"}")) + .unwrap(); + assert_eq!( + h.parts, + vec![ + input_start("call_1", "get_weather"), + input_delta("call_1", "{\"city\": \"London\"}"), + ] + ); + h.clear(); + + h.flush(); + assert_eq!( + h.parts, + vec![ + input_end("call_1"), + tool_call("call_1", "get_weather", "{\"city\": \"London\"}"), + ] + ); + } + + #[test] + fn does_not_finalize_when_argument_prefix_is_parsable_json() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "search", "{\"query\": \"test\"}")) + .unwrap(); + assert_eq!( + h.parts, + vec![ + input_start("call_1", "search"), + input_delta("call_1", "{\"query\": \"test\"}"), + ] + ); + + h.delta(cont(Some(0), ", \"limit\": 10}")).unwrap(); + h.flush(); + + assert_eq!( + h.parts.last(), + Some(&tool_call( + "call_1", + "search", + "{\"query\": \"test\"}, \"limit\": 10}" + )) + ); + assert_eq!(h.tool_calls().len(), 1); + } + + #[test] + fn multiple_concurrent_tool_calls() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "get_weather", "")).unwrap(); + h.delta(start(1, "call_2", "get_time", "")).unwrap(); + + assert_eq!( + h.parts, + vec![ + input_start("call_1", "get_weather"), + input_start("call_2", "get_time"), + ] + ); + } + + #[test] + fn non_zero_and_non_contiguous_indexes() { + let mut h = Harness::new(); + + h.delta(start(1, "call_1", "fn1", "{\"value\":1}")).unwrap(); + h.delta(start(3, "call_2", "fn2", "{\"value\":2}")).unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![ + tc("call_1", "fn1", "{\"value\":1}"), + tc("call_2", "fn2", "{\"value\":2}"), + ] + ); + } + + #[test] + fn keeps_distinct_tool_calls_that_reuse_an_index() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "fn", "{\"value\":1}")).unwrap(); + h.delta(start(0, "call_2", "fn", "{\"value\":2}")).unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![ + tc("call_1", "fn", "{\"value\":1}"), + tc("call_2", "fn", "{\"value\":2}"), + ] + ); + } + + #[test] + fn continues_latest_call_when_index_is_omitted_after_starting_at() { + for index in [None, Some(7)] { + let mut h = Harness::new(); + + h.delta(delta( + index, + Some("call_1"), + Some("function"), + Some("fn"), + Some("{\"val"), + )) + .unwrap(); + h.delta(cont(None, "ue\":1}")).unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![tc("call_1", "fn", "{\"value\":1}")], + "index {index:?}" + ); + } + } + + #[test] + fn uses_the_index_when_continuation_ids_are_empty() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "fn", "{\"val")).unwrap(); + h.delta(delta( + Some(0), + Some(""), + Some("function"), + None, + Some("ue\":1}"), + )) + .unwrap(); + h.flush(); + + assert_eq!(h.tool_calls(), vec![tc("call_1", "fn", "{\"value\":1}")]); + } + + #[test] + fn skips_deltas_for_already_finished_tool_calls() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "fn", "{}")).unwrap(); + h.flush(); + h.clear(); + + h.delta(cont(Some(0), "extra")).unwrap(); + assert_eq!(h.parts, vec![]); + } + + #[test] + fn skips_delta_emission_when_arguments_are_null() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "fn", "")).unwrap(); + h.clear(); + + h.delta(delta(Some(0), None, None, None, None)).unwrap(); + assert_eq!(h.parts, vec![]); + } + + #[test] + fn uses_index_fallback_when_index_is_not_provided() { + let mut h = Harness::new(); + + h.delta(delta( + None, + Some("call_1"), + Some("function"), + Some("fn1"), + Some("{}"), + )) + .unwrap(); + h.delta(delta( + None, + Some("call_2"), + Some("function"), + Some("fn2"), + Some("{}"), + )) + .unwrap(); + + let starts: Vec<&Part> = h + .parts + .iter() + .filter(|p| matches!(p, Part::ToolInputStart { .. })) + .collect(); + assert_eq!( + starts, + vec![&input_start("call_1", "fn1"), &input_start("call_2", "fn2")] + ); + } + + #[test] + fn generates_an_id_when_id_is_missing() { + let mut h = Harness::with( + StreamingToolCallTracker::new().with_generate_id(|| "generated-id".to_string()), + ); + + h.delta(delta( + Some(0), + None, + Some("function"), + Some("fn"), + Some("{}"), + )) + .unwrap(); + h.flush(); + + assert_eq!(h.tool_calls(), vec![tc("generated-id", "fn", "{}")]); + } + + #[test] + fn errors_when_function_name_is_missing() { + // TS covers `name: undefined` and `name: null`; both are `None`. + let mut h = Harness::new(); + + let result = h.delta(delta(Some(0), Some("call_1"), Some("function"), None, None)); + + assert_eq!(result, Err(TrackerError::MissingFunctionName)); + assert_eq!( + TrackerError::MissingFunctionName.to_string(), + "Expected 'function.name' to be a string." + ); + } + + #[test] + fn ignores_a_blank_function_name_without_preventing_prior_calls_from_finalizing() { + for name in ["", " "] { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "valid_tool", "{\"value\":1}")) + .unwrap(); + h.delta(start(1, "call_2", name, "{\"value\":2}")).unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![tc("call_1", "valid_tool", "{\"value\":1}")], + "name {name:?}" + ); + } + } + + #[test] + fn retains_continuation_arguments_for_a_blank_name() { + // (blank name, continuation id, continuation index) + let cases = [ + ("", Some("call_1"), None), // blank name with a matching id + (" ", None, Some(0)), // whitespace name, matching index + ]; + for (name, id, index) in cases { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "read_file", "{\"pa")).unwrap(); + h.delta(delta(index, id, None, Some(name), Some("th\":\"a\"}"))) + .unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![tc("call_1", "read_file", "{\"path\":\"a\"}")], + "name {name:?}" + ); + } + } + + #[test] + fn keeps_id_less_calls_distinct_when_an_index_is_reused_and_type_is_omitted() { + let (generate, _) = counting_generator(&["generated-1", "generated-2", "generated-3"]); + let mut h = Harness::with(StreamingToolCallTracker::new().with_generate_id(generate)); + + h.delta(delta( + Some(0), + None, + None, + Some("read_file"), + Some("{\"path\":\"p0\"}"), + )) + .unwrap(); + h.delta(delta( + Some(0), + None, + None, + Some("write_file"), + Some("{\"path\":\"p1\"}"), + )) + .unwrap(); + h.delta(delta( + Some(0), + None, + None, + Some("read_file"), + Some("{\"path\":\"p2\"}"), + )) + .unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![ + tc("generated-1", "read_file", "{\"path\":\"p0\"}"), + tc("generated-2", "write_file", "{\"path\":\"p1\"}"), + tc("generated-3", "read_file", "{\"path\":\"p2\"}"), + ] + ); + } + + #[test] + fn keeps_complete_same_name_calls_distinct_with_reused_index() { + for id in [None, Some("dup")] { + let mut h = Harness::with( + StreamingToolCallTracker::new().with_generate_id(|| "generated-id".to_string()), + ); + + for value in ["{\"value\":1}", "{\"value\":2}"] { + h.delta(delta( + Some(0), + id, + Some("function"), + Some("same_tool"), + Some(value), + )) + .unwrap(); + } + h.flush(); + + let calls = h.tool_calls(); + let inputs: Vec<&str> = calls.iter().map(|c| c.2.as_str()).collect(); + assert_eq!(inputs, vec!["{\"value\":1}", "{\"value\":2}"], "id {id:?}"); + let ids: std::collections::HashSet<&str> = calls.iter().map(|c| c.0.as_str()).collect(); + assert_eq!(ids.len(), 2, "id {id:?}"); + } + } + + #[test] + fn keeps_a_partial_same_name_call_distinct_with_reused_index() { + for id in [None, Some("dup")] { + let mut h = Harness::with( + StreamingToolCallTracker::new().with_generate_id(|| "generated-id".to_string()), + ); + + for value in ["{\"value\":1}", "{\"value\":"] { + h.delta(delta( + Some(0), + id, + Some("function"), + Some("same_tool"), + Some(value), + )) + .unwrap(); + } + h.flush(); + + let calls = h.tool_calls(); + let inputs: Vec<&str> = calls.iter().map(|c| c.2.as_str()).collect(); + assert_eq!(inputs, vec!["{\"value\":1}", "{\"value\":"], "id {id:?}"); + let ids: std::collections::HashSet<&str> = calls.iter().map(|c| c.0.as_str()).collect(); + assert_eq!(ids.len(), 2, "id {id:?}"); + } + } + + #[test] + fn keeps_interleaved_same_name_calls_with_distinct_ids_and_reused_index_separate() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "same_tool", "{\"value\":")) + .unwrap(); + h.delta(start(0, "call_2", "same_tool", "{\"value\":2}")) + .unwrap(); + h.delta(delta(Some(0), Some("call_1"), None, None, Some("1}"))) + .unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![ + tc("call_1", "same_tool", "{\"value\":1}"), + tc("call_2", "same_tool", "{\"value\":2}"), + ] + ); + } + + #[test] + fn ignores_an_index_only_continuation_after_the_index_is_reused() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "first", "{\"value\":1}")) + .unwrap(); + h.delta(start(0, "call_2", "second", "{\"value\":2}")) + .unwrap(); + h.delta(cont(Some(0), "{\"unattributed\":true}")).unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![ + tc("call_1", "first", "{\"value\":1}"), + tc("call_2", "second", "{\"value\":2}"), + ] + ); + } + + #[test] + fn uses_the_index_when_continuation_ids_are_blank() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "read_file", "{\"pa")).unwrap(); + h.delta(delta(Some(0), Some(" "), None, None, Some("th\":\"a\"}"))) + .unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![tc("call_1", "read_file", "{\"path\":\"a\"}")] + ); + } + + #[test] + fn generates_unique_ids_for_blank_and_repeated_ids() { + let (generate, _) = counting_generator(&["generated-1", "generated-2"]); + let mut h = Harness::with(StreamingToolCallTracker::new().with_generate_id(generate)); + + h.delta(start(0, "", "read_file", "{}")).unwrap(); + h.delta(start(1, "dup", "read_file", "{}")).unwrap(); + h.delta(start(2, "dup", "write_file", "{}")).unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![ + tc("generated-1", "read_file", "{}"), + tc("dup", "read_file", "{}"), + tc("generated-2", "write_file", "{}"), + ] + ); + } + + #[test] + fn keeps_same_name_calls_with_repeated_ids_and_distinct_indices_separate() { + let mut h = Harness::with( + StreamingToolCallTracker::new().with_generate_id(|| "generated-id".to_string()), + ); + + h.delta(start(0, "dup", "same_tool", "{\"value\":0}")) + .unwrap(); + h.delta(start(1, "dup", "same_tool", "{\"value\":1}")) + .unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![ + tc("dup", "same_tool", "{\"value\":0}"), + tc("generated-id", "same_tool", "{\"value\":1}"), + ] + ); + } + + #[test] + fn preserves_nonblank_ids_and_function_names_exactly() { + let mut h = Harness::new(); + + h.delta(start(0, " spaced ", " same_tool ", "{\"value\":")) + .unwrap(); + h.delta(delta(Some(0), Some(" spaced "), None, None, Some("0}"))) + .unwrap(); + h.delta(start(1, "spaced", " same_tool ", "{\"value\":1}")) + .unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![ + tc(" spaced ", " same_tool ", "{\"value\":0}"), + tc("spaced", " same_tool ", "{\"value\":1}"), + ] + ); + } + + #[test] + fn creates_bounded_unique_ids_when_generate_id_returns_duplicates() { + let (generate, calls) = counting_generator(&["generated-id"]); + let mut h = Harness::with(StreamingToolCallTracker::new().with_generate_id(generate)); + + for (index, name) in [(0, "first"), (1, "second"), (2, "third")] { + h.delta(delta( + Some(index), + None, + Some("function"), + Some(name), + Some("{}"), + )) + .unwrap(); + } + h.flush(); + + assert_eq!(calls.load(Ordering::SeqCst), 3); + let ids: Vec = h.tool_calls().into_iter().map(|c| c.0).collect(); + assert_eq!( + ids, + vec!["generated-id", "generated-id-1", "generated-id-2"] + ); + } + + #[test] + fn creates_usable_ids_when_generate_id_returns_blank_values() { + let (generate, calls) = counting_generator(&[" "]); + let mut h = Harness::with(StreamingToolCallTracker::new().with_generate_id(generate)); + + for (index, name) in [(0, "first"), (1, "second")] { + h.delta(delta( + Some(index), + None, + Some("function"), + Some(name), + Some("{}"), + )) + .unwrap(); + } + h.flush(); + + assert_eq!(calls.load(Ordering::SeqCst), 2); + let ids: Vec = h.tool_calls().into_iter().map(|c| c.0).collect(); + assert_eq!(ids, vec!["tool-call", "tool-call-1"]); + } + + #[test] + fn ignores_unattributable_deltas_when_multiple_calls_are_active() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "read_file", "{\"path\":\"a\"}")) + .unwrap(); + h.delta(start(1, "call_2", "write_file", "{\"path\":\"b\"}")) + .unwrap(); + h.delta(cont(None, "{\"unattributed\":true}")).unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![ + tc("call_1", "read_file", "{\"path\":\"a\"}"), + tc("call_2", "write_file", "{\"path\":\"b\"}"), + ] + ); + } + + #[test] + fn ignores_an_ambiguous_continuation_for_a_repeated_id() { + let mut h = Harness::with( + StreamingToolCallTracker::new().with_generate_id(|| "generated-id".to_string()), + ); + + h.delta(start(0, "dup", "read_file", "{\"path\":\"a\"}")) + .unwrap(); + h.delta(start(1, "dup", "write_file", "{\"path\":\"b\"}")) + .unwrap(); + h.delta(delta( + None, + Some("dup"), + None, + None, + Some("{\"unattributed\":true}"), + )) + .unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![ + tc("dup", "read_file", "{\"path\":\"a\"}"), + tc("generated-id", "write_file", "{\"path\":\"b\"}"), + ] + ); + } + + #[test] + fn uses_a_matching_name_and_index_for_an_id_less_continuation() { + let mut h = Harness::with( + StreamingToolCallTracker::new().with_type_validation(TypeValidation::Required), + ); + + h.delta(start(0, "call_1", "read_file", "{\"pa")).unwrap(); + h.delta(delta( + Some(0), + None, + None, + Some("read_file"), + Some("th\":\"a\"}"), + )) + .unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![tc("call_1", "read_file", "{\"path\":\"a\"}")] + ); + } + + #[test] + fn uses_the_index_when_a_continuation_has_an_unexpected_id() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "read_file", "{\"pa")).unwrap(); + h.delta(delta( + Some(0), + Some("unexpected"), + None, + None, + Some("th\":\"a\"}"), + )) + .unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![tc("call_1", "read_file", "{\"path\":\"a\"}")] + ); + } + + #[test] + fn continues_a_call_when_its_id_changes_but_index_and_name_match() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "read_file", "{\"pa")).unwrap(); + h.delta(start(0, "unexpected", "read_file", "th\":\"a\"}")) + .unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![tc("call_1", "read_file", "{\"path\":\"a\"}")] + ); + } + + #[test] + fn continues_a_call_when_all_labels_repeat_after_a_parsable_argument_prefix() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "calculate", "1")).unwrap(); + h.delta(start(0, "call_1", "calculate", "2")).unwrap(); + h.flush(); + + assert_eq!(h.tool_calls(), vec![tc("call_1", "calculate", "12")]); + } + + #[test] + fn continues_a_structured_argument_when_repeated_labels_precede_a_nested_object() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "calculate", "{\"value\":")) + .unwrap(); + h.delta(start(0, "call_1", "calculate", "{\"nested\":true}}")) + .unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![tc("call_1", "calculate", "{\"value\":{\"nested\":true}}")] + ); + } + + #[test] + fn emits_tool_calls_in_index_order() { + let mut h = Harness::new(); + + h.delta(start(1, "call_1", "second", "{}")).unwrap(); + h.delta(start(0, "call_0", "first", "{}")).unwrap(); + h.flush(); + + let names: Vec = h.tool_calls().into_iter().map(|c| c.1).collect(); + assert_eq!(names, vec!["first", "second"]); + } + + #[test] + fn preserves_insertion_order_when_calls_mix_present_and_omitted_indices() { + let mut h = Harness::new(); + + h.delta(delta( + None, + Some("call_without_index"), + Some("function"), + Some("first"), + Some("{}"), + )) + .unwrap(); + h.delta(start(0, "call_with_index", "second", "{}")) + .unwrap(); + h.flush(); + + let names: Vec = h.tool_calls().into_iter().map(|c| c.1).collect(); + assert_eq!(names, vec!["first", "second"]); + } + + #[test] + fn errors_when_function_name_is_missing_from_a_new_call() { + let mut h = Harness::new(); + + // `function: {}` — a function object with neither name nor arguments. + let result = h.delta(delta(Some(0), Some("call_1"), Some("function"), None, None)); + + assert_eq!(result, Err(TrackerError::MissingFunctionName)); + } +} + +mod type_validation { + use super::*; + + fn custom_type() -> StreamingToolCallDelta { + delta( + Some(0), + Some("call_1"), + Some("custom"), + Some("fn"), + Some(""), + ) + } + + fn no_type() -> StreamingToolCallDelta { + delta(Some(0), Some("call_1"), None, Some("fn"), Some("")) + } + + #[test] + fn does_not_validate_type_with_none() { + let mut h = Harness::with( + StreamingToolCallTracker::new().with_type_validation(TypeValidation::None), + ); + + assert_eq!(h.delta(custom_type()), Ok(())); + } + + #[test] + fn validates_type_when_present_with_if_present() { + let mut h = Harness::with( + StreamingToolCallTracker::new().with_type_validation(TypeValidation::IfPresent), + ); + + assert_eq!(h.delta(custom_type()), Err(TrackerError::InvalidType)); + assert_eq!( + TrackerError::InvalidType.to_string(), + "Expected 'function' type." + ); + + // A missing type is accepted. + assert_eq!(h.delta(no_type()), Ok(())); + } + + #[test] + fn requires_function_type_with_required() { + let mut h = Harness::with( + StreamingToolCallTracker::new().with_type_validation(TypeValidation::Required), + ); + + assert_eq!(h.delta(no_type()), Err(TrackerError::InvalidType)); + + assert_eq!(h.delta(start(0, "call_1", "fn", "")), Ok(())); + } +} + +mod flush { + use super::*; + + #[test] + fn finalizes_unfinished_tool_calls() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "fn", "{\"key\": \"val")) + .unwrap(); + h.clear(); + + h.flush(); + + assert_eq!( + h.parts, + vec![ + input_end("call_1"), + tool_call("call_1", "fn", "{\"key\": \"val"), + ] + ); + } + + #[test] + fn does_not_re_finalize_already_finished_tool_calls() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "fn", "{}")).unwrap(); + h.flush(); + h.clear(); + + h.flush(); + + assert_eq!(h.parts, vec![]); + } +} + +mod metadata { + use super::*; + + fn google_tracker() -> StreamingToolCallTracker { + StreamingToolCallTracker::new() + .with_extract_metadata(|delta| { + delta.extra["extra_content"]["google"]["thought_signature"] + .as_str() + .map(|sig| json!({ "thoughtSignature": sig })) + }) + .with_build_provider_metadata(|metadata| { + metadata + .and_then(|m| m.get("thoughtSignature")) + .map(|sig| json!({ "google": { "thoughtSignature": sig } })) + }) + } + + #[test] + fn extracts_and_includes_provider_metadata_in_tool_call_parts() { + let mut h = Harness::with_meta(google_tracker()); + + h.delta( + start(0, "call_1", "fn", "{}") + .extra(json!({ "extra_content": { "google": { "thought_signature": "sig123" } } })), + ) + .unwrap(); + h.flush(); + + let tool_call = h + .parts + .iter() + .find(|p| matches!(p, ToolCallStreamPart::ToolCall { .. })); + assert_eq!( + tool_call, + Some(&ToolCallStreamPart::ToolCall { + tool_call_id: "call_1".into(), + tool_name: "fn".into(), + input: "{}".into(), + provider_metadata: Some(json!({ "google": { "thoughtSignature": "sig123" } })), + }) + ); + } + + #[test] + fn includes_provider_metadata_for_unfinished_tool_calls_finalized_in_flush() { + let mut h = Harness::with_meta( + StreamingToolCallTracker::new() + .with_extract_metadata(|_| Some(json!({ "custom": { "key": "value" } }))) + .with_build_provider_metadata(|metadata| { + metadata.map(|m| json!({ "provider": m })) + }), + ); + + h.delta(start(0, "call_1", "fn", "{\"incomplete")).unwrap(); + h.clear(); + + h.flush(); + + assert_eq!( + h.parts.last(), + Some(&ToolCallStreamPart::ToolCall { + tool_call_id: "call_1".into(), + tool_name: "fn".into(), + input: "{\"incomplete".into(), + provider_metadata: Some(json!({ "provider": { "custom": { "key": "value" } } })), + }) + ); + } + + #[test] + fn omits_provider_metadata_when_the_builder_returns_none() { + let mut h = Harness::with_meta( + StreamingToolCallTracker::<()>::new() + .with_extract_metadata(|_| None) + .with_build_provider_metadata(|_| None), + ); + + h.delta(start(0, "call_1", "fn", "{}")).unwrap(); + h.flush(); + + let tool_call = h + .parts + .iter() + .find(|p| matches!(p, ToolCallStreamPart::ToolCall { .. })); + assert_eq!(tool_call, Some(&super::tool_call("call_1", "fn", "{}"))); + } +} + +mod generate_id { + use super::*; + + #[test] + fn keeps_the_wire_id_when_present() { + let mut h = Harness::with( + StreamingToolCallTracker::new().with_generate_id(|| "custom-id".to_string()), + ); + + h.delta(start(0, "call_1", "fn", "{\"key\": \"val")) + .unwrap(); + h.clear(); + + h.flush(); + + let ids: Vec = h.tool_calls().into_iter().map(|c| c.0).collect(); + assert_eq!(ids, vec!["call_1"]); + } +} From b17aa5d2ef897b4428d9dd1064c1c05fe995df60 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 05:20:50 +0000 Subject: [PATCH 3/9] feat(stream)!: align SseStream with eventsource-parser The AI SDK's parseJsonEventStream pipes through eventsource-parser. Replace the blank-line frame splitter with its line-based state machine: - \n, \r and \r\n line endings in any mix; a trailing \r waits for a possible \n in the next chunk - strip a UTF-8 BOM at the start of the stream - a field line without ':' has an empty value; empty event -> None; id containing U+0000 ignored; retry accepted only when all ASCII digits - exceeding the buffer limit (event data + partial line) is fatal and ends the stream, as maxBufferSize does upstream Kept from aimux: strict UTF-8 decoding after reassembly (an invalid line now drops only the event being built) and a default 1 MiB bound. extract_lines already matched upstream; the [DONE]-skipping JSON layer in aimux-provider-utils already matches parseJsonEventStream. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_015EYDYDWYcPsDmjBjFeuDVe --- CHANGELOG.md | 9 + aimux-stream/src/sse.rs | 372 ++++++++++++++++++++------------- aimux-stream/tests/sse_test.rs | 184 +++++++++++++++- 3 files changed, 415 insertions(+), 150 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 288c76a2..d9b6ba04 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -24,6 +24,15 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 be `Send + Sync`. - Added `StreamingToolCallArgumentState` and `starts_with_structured_value`, the structural JSON-prefix tracker the new correlation logic relies on. +- `SseStream` rewritten as a line-based parser matching `eventsource-parser` + (the parser behind the AI SDK's `parseJsonEventStream`): `\n`, `\r` and + `\r\n` line endings in any mix, a leading UTF-8 BOM is stripped, a field + line without `:` counts as an empty value, an empty `event:` is `None`, an + `id` containing U+0000 is ignored, and `retry` must be all ASCII digits. + Exceeding `max_event_size` (now: buffered event data plus the partial line) + yields `SseError::FrameTooLarge` and ends the stream instead of skipping the + frame. Invalid UTF-8 still yields `SseError::Utf8`, now dropping only the + event being built. ## [0.5.0] - 2026-09-27 diff --git a/aimux-stream/src/sse.rs b/aimux-stream/src/sse.rs index c53f66e4..73eb360f 100644 --- a/aimux-stream/src/sse.rs +++ b/aimux-stream/src/sse.rs @@ -1,4 +1,28 @@ //! SSE (Server-Sent Events) parser for streaming model responses. +//! +//! Line-based incremental parser following `eventsource-parser`'s +//! `createParser` (the parser behind the AI SDK's `parseJsonEventStream`) and +//! the WHATWG "parsing an event stream" algorithm: +//! +//! - lines end in `\n`, `\r` or `\r\n`, freely mixed; a `\r` at the end of +//! the buffered input waits for the next chunk in case a `\n` follows; +//! - a UTF-8 BOM at the very start of the stream is stripped; +//! - a blank line dispatches the event, but only if it had at least one +//! `data` line; field state resets after every blank line; +//! - `field: value` removes exactly one leading U+0020 SPACE; a line without +//! `:` is a field with an empty value; `:`-prefixed lines are comments; +//! - an empty `event` value means no event type; an `id` containing U+0000 is +//! ignored; `retry` is accepted only when it is all ASCII digits; unknown +//! fields are ignored; +//! - a partial event at end of stream is not dispatched. +//! +//! Two aimux additions: every line is strictly UTF-8 decoded after it has +//! been reassembled (a code point split across chunks is never corrupted, and +//! invalid UTF-8 surfaces as [`SseError::Utf8`] and drops the event being +//! built), and buffered input is bounded by `max_event_size` (the parser's +//! `maxBufferSize`; exceeding it is fatal and ends the stream, as upstream). + +use std::collections::VecDeque; use bytes::Bytes; use futures::Stream; @@ -7,9 +31,11 @@ use std::pin::Pin; use std::task::{Context, Poll}; use thiserror::Error; -/// Default upper bound on a single buffered SSE event's size (1 MiB). +/// Default upper bound on buffered event data plus any partial line (1 MiB). const DEFAULT_MAX_EVENT_SIZE: usize = 1024 * 1024; +const BOM: &[u8] = b"\xEF\xBB\xBF"; + #[derive(Debug, Error)] pub enum SseError { #[error("utf-8 decode error: {0}")] @@ -23,30 +49,188 @@ pub enum SseError { /// A parsed SSE event. #[derive(Debug, Clone, Default)] pub struct SseEvent { - /// The `event:` field (optional). + /// The `event:` field (`None` when absent or empty). pub event: Option, - /// The `data:` field. + /// The `data:` field; multiple `data` lines are joined with `\n`. pub data: String, /// The `id:` field (optional). pub id: Option, - /// The `retry:` field (optional). + /// The `retry:` field (optional; all-digit values only). pub retry: Option, } +/// Incremental SSE line parser (the `createParser` state machine). +#[derive(Debug)] +struct Parser { + /// Bytes after the last complete line terminator. + pending: Vec, + /// Offset in `pending` from which the next terminator search starts, so a + /// long line fed in many small chunks is scanned once, not quadratically. + scan_from: usize, + bom_checked: bool, + event: Option, + data: String, + data_lines: usize, + id: Option, + retry: Option, + max_size: usize, + terminated: bool, + ready: VecDeque>, +} + +impl Parser { + fn new(max_size: usize) -> Self { + Self { + pending: Vec::new(), + scan_from: 0, + bom_checked: false, + event: None, + data: String::new(), + data_lines: 0, + id: None, + retry: None, + max_size, + terminated: false, + ready: VecDeque::new(), + } + } + + fn feed(&mut self, chunk: &[u8]) { + if self.terminated { + return; + } + self.pending.extend_from_slice(chunk); + + if !self.bom_checked { + // Wait until the first three bytes can rule a BOM in or out. + if self.pending.len() < BOM.len() && BOM.starts_with(&self.pending) { + return; + } + if self.pending.starts_with(BOM) { + self.pending.drain(..BOM.len()); + } + self.bom_checked = true; + } + + let mut line_start = 0; + let mut search = self.scan_from; + while let Some(offset) = self.pending[search..] + .iter() + .position(|&b| b == b'\n' || b == b'\r') + { + let end = search + offset; + let is_cr = self.pending[end] == b'\r'; + // A trailing `\r` may be the first half of a `\r\n` split across + // chunks: defer it until more input arrives. + if is_cr && end + 1 == self.pending.len() { + break; + } + let line = self.pending[line_start..end].to_vec(); + self.parse_line(&line); + if self.terminated { + return; + } + line_start = end + 1; + if is_cr && self.pending.get(line_start) == Some(&b'\n') { + line_start += 1; + } + search = line_start; + } + self.pending.drain(..line_start); + // Resume the next search at the deferred `\r`, if any. + self.scan_from = self.pending.len().saturating_sub(1); + if self.pending.last() != Some(&b'\r') { + self.scan_from = self.pending.len(); + } + self.check_size(self.pending.len()); + } + + fn parse_line(&mut self, line: &[u8]) { + if line.is_empty() { + self.dispatch(); + return; + } + let line = match String::from_utf8(line.to_vec()) { + Ok(line) => line, + Err(error) => { + // The event being built can no longer be trusted. + self.reset_event(); + self.ready.push_back(Err(SseError::Utf8(error))); + return; + } + }; + if line.starts_with(':') { + return; // comment + } + let (field, value) = match line.split_once(':') { + Some((field, value)) => (field, value.strip_prefix(' ').unwrap_or(value)), + None => (line.as_str(), ""), + }; + match field { + "event" => self.event = (!value.is_empty()).then(|| value.to_string()), + "data" => { + if self.data_lines > 0 { + self.data.push('\n'); + } + self.data.push_str(value); + self.data_lines += 1; + // Lines already consumed from `pending` are not buffered any + // more; only the event data counts until the chunk is done. + self.check_size(0); + } + "id" => { + if !value.contains('\0') { + self.id = Some(value.to_string()); + } + } + "retry" => { + if !value.is_empty() && value.bytes().all(|b| b.is_ascii_digit()) { + self.retry = value.parse().ok(); + } + } + _ => {} // unknown field + } + } + + fn dispatch(&mut self) { + if self.data_lines > 0 { + self.ready.push_back(Ok(SseEvent { + event: self.event.take(), + data: std::mem::take(&mut self.data), + id: self.id.take(), + retry: self.retry.take(), + })); + } + self.reset_event(); + } + + fn reset_event(&mut self) { + self.event = None; + self.data.clear(); + self.data_lines = 0; + self.id = None; + self.retry = None; + } + + /// `maxBufferSize`: buffered event data plus the partial line. Exceeding + /// it is fatal, as in `EventSourceParserStream`. + fn check_size(&mut self, pending_len: usize) { + if self.terminated || pending_len + self.data.len() <= self.max_size { + return; + } + self.terminated = true; + self.pending.clear(); + self.reset_event(); + self.ready.push_back(Err(SseError::FrameTooLarge)); + } +} + pin_project! { /// An adapter that decodes a byte stream into SSE events. - /// - /// Bytes are accumulated in a raw `Vec` buffer and split on the SSE - /// event terminator (a blank line). Each complete frame is strictly - /// UTF-8 decoded only *after* reassembly, so a multi-byte character split - /// across two network chunks is never corrupted into replacement chars — - /// unlike a per-chunk `String::from_utf8_lossy` decode, which would emit a - /// `U+FFFD` on each side of the split. pub struct SseStream { #[pin] inner: S, - buffer: Vec, - max_event_size: usize, + parser: Parser, done: bool, _err: std::marker::PhantomData, } @@ -60,16 +244,13 @@ where Self::with_max_event_size(stream, DEFAULT_MAX_EVENT_SIZE) } - /// Create an [`SseStream`] with a custom per-event size limit. A single - /// event frame (the bytes between two terminators, excluding the - /// terminator itself) larger than `max_event_size` bytes yields - /// [`SseError::FrameTooLarge`], as does a buffer that grows past the limit - /// while waiting for a terminator. + /// Create an [`SseStream`] with a custom buffer limit: when buffered event + /// data plus the current partial line exceed `max_event_size` bytes, the + /// stream yields [`SseError::FrameTooLarge`] and ends. pub fn with_max_event_size(stream: S, max_event_size: usize) -> Self { Self { inner: stream, - buffer: Vec::new(), - max_event_size, + parser: Parser::new(max_event_size), done: false, _err: std::marker::PhantomData, } @@ -87,163 +268,58 @@ where let this = self.as_mut().get_mut(); loop { - // Try to emit a complete event from the buffer. - if let Some((frame_len, sep_len)) = find_separator(&this.buffer) { - if frame_len > this.max_event_size { - // Drop the oversized frame so a retried poll makes progress. - this.buffer.drain(..frame_len + sep_len); - return Poll::Ready(Some(Err(SseError::FrameTooLarge))); - } - // Extract the frame bytes and drop the terminator. Decoding the - // fully reassembled frame with `String::from_utf8` (rather than - // `from_utf8_lossy` per chunk) preserves code points split - // across chunks and surfaces invalid UTF-8 as an error. - let frame_bytes: Vec = this.buffer.drain(..frame_len).collect(); - this.buffer.drain(..sep_len); - let frame = match String::from_utf8(frame_bytes) { - Ok(s) => s, - Err(e) => return Poll::Ready(Some(Err(SseError::Utf8(e)))), - }; - let (event, has_data_line) = try_parse_event(&frame); - // Per the SSE spec / eventsource-parser, an event is dispatched - // only if it had at least one `data:` line (dataLines > 0). - // This covers comment-only, `event:`/`id:`/`retry:`-only, and - // blank-line keep-alives (all dispatched as nothing), while an - // explicit empty data line (`data:\n\n`) is still dispatched. - if !has_data_line { - continue; - } - return Poll::Ready(Some(Ok(event))); - } - - // No complete event yet. Guard against unbounded buffer growth when - // a terminator never arrives. The partial can never form a valid - // frame under the limit, so drop it — a retried poll then makes - // progress instead of looping on the same oversized buffer. - if this.buffer.len() > this.max_event_size { - this.buffer.clear(); - return Poll::Ready(Some(Err(SseError::FrameTooLarge))); + if let Some(item) = this.parser.ready.pop_front() { + return Poll::Ready(Some(item)); } - - if this.done { - // No terminating blank line: the buffered partial is not a - // complete event and is dropped. This matches eventsource-parser's - // `EventSourceParserStream`, which has no flush handler — a - // partial event at EOF is simply not dispatched. + // A partial event at end of stream is not dispatched: the + // upstream stream has no flush handler. + if this.done || this.parser.terminated { return Poll::Ready(None); } - - // Read more data. match Pin::new(&mut this.inner).poll_next(cx) { - Poll::Ready(Some(Ok(bytes))) => { - this.buffer.extend_from_slice(&bytes); - } + Poll::Ready(Some(Ok(bytes))) => this.parser.feed(&bytes), Poll::Ready(Some(Err(e))) => { return Poll::Ready(Some(Err(SseError::Stream(e.to_string())))); } - Poll::Ready(None) => { - this.done = true; - } + Poll::Ready(None) => this.done = true, Poll::Pending => return Poll::Pending, } } } } -/// Locate the next SSE event terminator (a blank line) in `buf`. -/// -/// Returns `(frame_len, sep_len)` where `frame_len` is the number of bytes -/// *before* the terminator and `sep_len` is the terminator's length. Prefers -/// `\n\n` (so a pure-CRLF stream — which contains no bare `\n\n` — falls -/// through to `\r\n\r\n`), mirroring the str-based -/// `find("\n\n").or_else(|| find("\r\n\r\n"))`. -fn find_separator(buf: &[u8]) -> Option<(usize, usize)> { - if let Some(pos) = find_subsequence(buf, b"\n\n") { - return Some((pos, 2)); - } - find_subsequence(buf, b"\r\n\r\n").map(|pos| (pos, 4)) -} - -/// First index of `needle` in `haystack`, comparing raw bytes. -fn find_subsequence(haystack: &[u8], needle: &[u8]) -> Option { - haystack.windows(needle.len()).position(|w| w == needle) -} - -/// Parse a single complete SSE event frame (the bytes between two -/// terminators, already strictly UTF-8 decoded) into an [`SseEvent`] plus a -/// flag indicating whether the event contained at least one `data:` line. -/// -/// A complete event is terminated by a blank line: `\n\n` or `\r\n\r\n` (the -/// terminator is consumed by the caller before this runs). Per the SSE spec, -/// exactly one leading U+0020 SPACE after the `:` is removed from each field -/// value (not all leading whitespace). Comment lines (starting with `:`) and -/// unknown fields are ignored. -fn try_parse_event(frame: &str) -> (SseEvent, bool) { - let mut event = SseEvent::default(); - let mut has_data_line = false; - for line in frame.lines() { - if let Some(value) = field_value(line, "data:") { - has_data_line = true; - if event.data.is_empty() { - event.data = value.to_string(); - } else { - event.data.push('\n'); - event.data.push_str(value); - } - } else if let Some(value) = field_value(line, "event:") { - event.event = Some(value.to_string()); - } else if let Some(value) = field_value(line, "id:") { - event.id = Some(value.to_string()); - } else if let Some(value) = field_value(line, "retry:") { - event.retry = value.parse().ok(); - } - // Comment lines (`:` prefix) and unknown fields are ignored. - } - - (event, has_data_line) -} - -/// Strip `prefix` from `line` and remove exactly one leading U+0020 SPACE from -/// the remainder (per the SSE spec). Returns `None` if `line` does not start -/// with `prefix`. -fn field_value<'a>(line: &'a str, prefix: &str) -> Option<&'a str> { - let rest = line.strip_prefix(prefix)?; - // Remove exactly one leading space, if present. - Some(rest.strip_prefix(' ').unwrap_or(rest)) -} - #[cfg(test)] mod tests { use super::*; + fn parse(input: &str) -> Vec { + let mut parser = Parser::new(DEFAULT_MAX_EVENT_SIZE); + parser.feed(input.as_bytes()); + parser.ready.into_iter().map(Result::unwrap).collect() + } + #[test] fn parse_single_event() { - let (event, has_data_line) = try_parse_event("data: hello world"); - assert_eq!(event.data, "hello world"); - assert!(has_data_line); + assert_eq!(parse("data: hello world\n\n")[0].data, "hello world"); } #[test] fn parse_multi_line_data() { - let (event, has_data_line) = try_parse_event("data: line1\ndata: line2"); - assert_eq!(event.data, "line1\nline2"); - assert!(has_data_line); + assert_eq!( + parse("data: line1\ndata: line2\n\n")[0].data, + "line1\nline2" + ); } #[test] fn parse_event_with_type() { - let (event, has_data_line) = try_parse_event("event: message\ndata: payload"); - assert_eq!(event.event.as_deref(), Some("message")); - assert_eq!(event.data, "payload"); - assert!(has_data_line); + let events = parse("event: message\ndata: payload\n\n"); + assert_eq!(events[0].event.as_deref(), Some("message")); + assert_eq!(events[0].data, "payload"); } #[test] - fn parse_event_without_data_has_no_data_line() { - // An `event:`-only event has no data line -> not dispatched. - let (event, has_data_line) = try_parse_event("event: ping"); - assert_eq!(event.event.as_deref(), Some("ping")); - assert!(event.data.is_empty()); - assert!(!has_data_line); + fn parse_event_without_data_is_not_dispatched() { + assert!(parse("event: ping\n\n").is_empty()); } } diff --git a/aimux-stream/tests/sse_test.rs b/aimux-stream/tests/sse_test.rs index 65e8f344..58be9a19 100644 --- a/aimux-stream/tests/sse_test.rs +++ b/aimux-stream/tests/sse_test.rs @@ -6,12 +6,17 @@ //! //! The expected behavior mirrors `eventsource-parser`'s `EventSourceParserStream` //! (which the TS `parseJsonEventStream` pipes through): -//! - an event is dispatched only on a terminating blank line (`\n\n`/`\r\n\r\n`); +//! - lines end in `\n`, `\r` or `\r\n` (freely mixed); an event is +//! dispatched only on a blank line; //! - an event with no `data:` line is NOT dispatched (covers comment-only, //! `event:`/`id:`/`retry:`-only, and blank-line keep-alives); //! - exactly one leading U+0020 SPACE after the `:` is stripped from a field //! value (per the SSE spec); -//! - a partial event at EOF (no terminating blank line) is dropped. +//! - a partial event at EOF (no terminating blank line) is dropped; +//! - a UTF-8 BOM at stream start is stripped; a line without `:` is a field +//! with an empty value; an empty `event` means no type; an `id` with +//! U+0000 is ignored; `retry` must be all ASCII digits; +//! - exceeding the buffer limit is fatal and ends the stream. use aimux_stream::{SseError, SseEvent, SseStream}; use bytes::Bytes; @@ -353,3 +358,178 @@ async fn buffer_growing_past_limit_without_terminator_returns_frame_too_large() assert_eq!(results.len(), 1); assert!(matches!(results[0], Err(SseError::FrameTooLarge))); } + +// ── eventsource-parser alignment ───────────────────────────────────────── + +#[tokio::test] +async fn bare_cr_line_endings() { + let events = collect_events(vec![ + "event: a\rdata: one\rdata: two\r\rdata: next\r\r:keep-alive\n", + ]) + .await; + assert_eq!(events.len(), 2); + let first = events[0].as_ref().unwrap(); + assert_eq!(first.event.as_deref(), Some("a")); + assert_eq!(data(first), "one\ntwo"); + assert_eq!(data(events[1].as_ref().unwrap()), "next"); +} + +#[tokio::test] +async fn trailing_cr_at_end_of_stream_does_not_dispatch() { + // A `\r` at the end of the buffered input waits for a possible `\n`; with + // no flush at end of stream (as `EventSourceParserStream`) the event is + // never completed. + let events = collect_events(vec!["data: a\r\rdata: b\r\r"]).await; + assert_eq!(events.len(), 1); + assert_eq!(data(events[0].as_ref().unwrap()), "a"); +} + +#[tokio::test] +async fn mixed_terminators_within_one_event() { + // `\r\n` then a bare `\n` blank line: one event, as eventsource-parser. + let events = collect_events(vec!["data: a\r\ndata: b\n\ndata: c\r\r\n"]).await; + assert_eq!(events.len(), 2); + assert_eq!(data(events[0].as_ref().unwrap()), "a\nb"); + assert_eq!(data(events[1].as_ref().unwrap()), "c"); +} + +#[tokio::test] +async fn crlf_split_between_chunks_is_one_terminator() { + // A `\r` at the end of a chunk must wait: the `\n` in the next chunk + // completes the same terminator, not an extra blank line. + let events = collect_events(vec!["data: a\r", "\ndata: b\r", "\n\r", "\n"]).await; + assert_eq!(events.len(), 1); + assert_eq!(data(events[0].as_ref().unwrap()), "a\nb"); +} + +#[tokio::test] +async fn leading_bom_is_stripped() { + let events = collect_events(vec!["\u{FEFF}data: first\n\n"]).await; + assert_eq!(events.len(), 1); + assert_eq!(data(events[0].as_ref().unwrap()), "first"); +} + +#[tokio::test] +async fn leading_bom_split_across_chunks_is_stripped() { + let bom = "\u{FEFF}".as_bytes(); + let events = collect_events_bytes(vec![ + bom[..1].to_vec(), + bom[1..].to_vec(), + b"data: first\n\n".to_vec(), + ]) + .await; + assert_eq!(events.len(), 1); + assert_eq!(data(events[0].as_ref().unwrap()), "first"); +} + +#[tokio::test] +async fn bom_after_stream_start_is_not_stripped() { + // Only the very start of the stream may carry a BOM; later it is part of + // the field name, which is then unknown. + let events = collect_events(vec!["data: a\n\n\u{FEFF}data: b\n\n"]).await; + assert_eq!(events.len(), 1); + assert_eq!(data(events[0].as_ref().unwrap()), "a"); +} + +#[tokio::test] +async fn field_without_colon_has_empty_value() { + // `data` alone is a data line with an empty value. + let events = collect_events(vec!["data\n\ndata\ndata: x\n\n"]).await; + assert_eq!(events.len(), 2); + assert_eq!(data(events[0].as_ref().unwrap()), ""); + assert_eq!(data(events[1].as_ref().unwrap()), "\nx"); +} + +#[tokio::test] +async fn empty_event_field_means_no_event_type() { + let events = collect_events(vec!["event:\ndata: x\n\n"]).await; + assert_eq!(events[0].as_ref().unwrap().event, None); +} + +#[tokio::test] +async fn event_type_resets_after_each_dispatch() { + let events = collect_events(vec!["event: a\ndata: 1\n\ndata: 2\n\n"]).await; + assert_eq!(events[0].as_ref().unwrap().event.as_deref(), Some("a")); + assert_eq!(events[1].as_ref().unwrap().event, None); +} + +#[tokio::test] +async fn id_containing_null_is_ignored() { + let events = collect_events(vec!["id: ok\nid: bad\u{0}id\ndata: x\n\n"]).await; + assert_eq!(events[0].as_ref().unwrap().id.as_deref(), Some("ok")); +} + +#[tokio::test] +async fn retry_must_be_all_ascii_digits() { + // `retry: 5` loses its one leading space and is valid; a second space is not. + for value in ["+5", "-5", "5s", " 5", ""] { + let input = format!("retry:{value}\ndata: x\n\n"); + let events = collect_events(vec![input.as_str()]).await; + assert_eq!(events[0].as_ref().unwrap().retry, None, "retry {value:?}"); + } + let events = collect_events(vec!["retry:1500\ndata: x\n\n"]).await; + assert_eq!(events[0].as_ref().unwrap().retry, Some(1500)); +} + +#[tokio::test] +async fn field_names_are_case_sensitive() { + let events = collect_events(vec!["Data: x\n\ndata: y\n\n"]).await; + assert_eq!(events.len(), 1); + assert_eq!(data(events[0].as_ref().unwrap()), "y"); +} + +#[tokio::test] +async fn invalid_utf8_drops_only_the_current_event() { + let mut bytes = b"data: ".to_vec(); + bytes.extend_from_slice(&[0xE4, 0xBD]); + bytes.extend_from_slice(b"\ndata: tail\n\ndata: next\n\n"); + let events = collect_events_bytes(vec![bytes]).await; + assert_eq!(events.len(), 3); + assert!(matches!(events[0], Err(SseError::Utf8(_)))); + // The line after the bad one starts a fresh event buffer. + assert_eq!(data(events[1].as_ref().unwrap()), "tail"); + assert_eq!(data(events[2].as_ref().unwrap()), "next"); +} + +#[tokio::test] +async fn exceeding_the_limit_ends_the_stream() { + let stream = SseStream::with_max_event_size( + stream::iter(vec![ + Ok::<_, std::io::Error>(Bytes::from_static(b"data: ok\n\n")), + Ok(Bytes::from_static(b"data: far too long for the limit\n\n")), + Ok(Bytes::from_static(b"data: ok\n\n")), + ]), + 10, + ); + let results: Vec<_> = stream.collect().await; + assert_eq!(results.len(), 2); + assert_eq!(data(results[0].as_ref().unwrap()), "ok"); + assert!(matches!(results[1], Err(SseError::FrameTooLarge))); +} + +#[tokio::test] +async fn multi_event_payload_split_at_every_byte_boundary() { + let payload = "event: e\r\nid: 1\r\ndata: 你好\r\n\r\ndata: a\rdata: b\r\r:c\ndata: z\n\n"; + let bytes = payload.as_bytes(); + let expected = collect_events(vec![payload]).await; + let expected: Vec<(Option, String, Option)> = expected + .into_iter() + .map(|e| { + let e = e.unwrap(); + (e.event, e.data, e.id) + }) + .collect(); + assert_eq!(expected.len(), 3); + for split in 1..bytes.len() { + let (a, b) = bytes.split_at(split); + let got: Vec<_> = collect_events_bytes(vec![a.to_vec(), b.to_vec()]) + .await + .into_iter() + .map(|e| { + let e = e.unwrap_or_else(|err| panic!("split {split}: {err:?}")); + (e.event, e.data, e.id) + }) + .collect(); + assert_eq!(got, expected, "split at byte {split}"); + } +} From 4e06914f252f37c0d5e1eb31aa7ff551399f0d66 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 05:22:09 +0000 Subject: [PATCH 4/9] refactor(stream)!: remove unused NdjsonStream Nothing in the workspace used NdjsonStream / NdjsonError and the AI SDK's provider-utils has no counterpart. Also move tokio to dev-dependencies (only the tests use it), drop the unused direct serde dependency, and update the crate description in README / PROJECT-OVERVIEW. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_015EYDYDWYcPsDmjBjFeuDVe --- CHANGELOG.md | 3 + README.md | 4 +- aimux-stream/Cargo.toml | 7 +- aimux-stream/src/lib.rs | 5 +- aimux-stream/src/ndjson.rs | 146 ------------------------------ aimux-stream/tests/ndjson_test.rs | 98 -------------------- docs/PROJECT-OVERVIEW.md | 2 +- 7 files changed, 12 insertions(+), 253 deletions(-) delete mode 100644 aimux-stream/src/ndjson.rs delete mode 100644 aimux-stream/tests/ndjson_test.rs diff --git a/CHANGELOG.md b/CHANGELOG.md index d9b6ba04..3cb55c7e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -33,6 +33,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 yields `SseError::FrameTooLarge` and ends the stream instead of skipping the frame. Invalid UTF-8 still yields `SseError::Utf8`, now dropping only the event being built. +- Removed `NdjsonStream` / `NdjsonError`: nothing in the workspace used them + and the AI SDK has no counterpart. `tokio` is now a dev-dependency only and + the unused direct `serde` dependency is dropped. ## [0.5.0] - 2026-09-27 diff --git a/README.md b/README.md index e636bc1b..02ee0ed0 100644 --- a/README.md +++ b/README.md @@ -98,7 +98,7 @@ middleware, and telemetry per request). aimux/ ├── aimux-core # Core abstractions: LanguageModel / Provider / Message / StreamPart ├── aimux-providers # Provider implementations — registry-backed + typed (docs/api/providers.md) -├── aimux-stream # SSE / NDJSON stream parsing +├── aimux-stream # SSE parsing, streamed tool-call tracking ├── aimux-provider-utils # One-exchange HTTP helpers, response handlers, API-key loading ├── aimux-ffi # C ABI (opaque handles + JSON results + owned aimux_error_t *) for non-native bindings └── tools/ # aimux-cli (cache probe) · aimux-replay · aimux-web (console) @@ -122,7 +122,7 @@ cargo add aimux-core aimux-providers |-------|-------------|-----------| | `aimux-core` | Core abstractions: `LanguageModel` / `Provider` / `Message` / `StreamPart` | [crates.io](https://crates.io/crates/aimux-core) | | `aimux-providers` | Provider implementations — [registry-backed + typed](docs/api/providers.md) | [crates.io](https://crates.io/crates/aimux-providers) | -| `aimux-stream` | SSE / NDJSON stream parsing | [crates.io](https://crates.io/crates/aimux-stream) | +| `aimux-stream` | SSE parsing, streamed tool-call tracking | [crates.io](https://crates.io/crates/aimux-stream) | | `aimux-provider-utils` | One-exchange HTTP helpers and typed response handlers | [crates.io](https://crates.io/crates/aimux-provider-utils) | | `aimux-ffi` | C ABI for non-native bindings | [crates.io](https://crates.io/crates/aimux-ffi) | diff --git a/aimux-stream/Cargo.toml b/aimux-stream/Cargo.toml index 6ee65c24..7c65cfec 100644 --- a/aimux-stream/Cargo.toml +++ b/aimux-stream/Cargo.toml @@ -4,18 +4,19 @@ version.workspace = true edition.workspace = true license.workspace = true description = "Streaming primitives for aimux" -keywords = ["llm", "streaming", "sse", "ndjson"] +keywords = ["llm", "streaming", "sse", "tool-calls"] categories = ["asynchronous", "parser-implementations"] documentation = "https://docs.rs/aimux-stream" [dependencies] futures = { workspace = true } -tokio = { workspace = true } pin-project-lite = { workspace = true } bytes = { workspace = true } -serde = { workspace = true } serde_json = { workspace = true } thiserror = { workspace = true } +[dev-dependencies] +tokio = { workspace = true } + [lints] workspace = true diff --git a/aimux-stream/src/lib.rs b/aimux-stream/src/lib.rs index d506e0ad..d9078d8d 100644 --- a/aimux-stream/src/lib.rs +++ b/aimux-stream/src/lib.rs @@ -1,17 +1,16 @@ //! # aimux-stream //! -//! Low-level streaming primitives for SSE (Server-Sent Events) and NDJSON parsing, +//! Low-level streaming primitives for SSE (Server-Sent Events) parsing and +//! streamed tool-call tracking, //! used by provider implementations to decode model API response streams. pub mod lines; -pub mod ndjson; pub mod sse; pub mod streaming_tool_call_argument_state; pub mod streaming_tool_call_tracker; // Re-export the most commonly used items. pub use lines::extract_lines; -pub use ndjson::{NdjsonError, NdjsonStream}; pub use sse::{SseError, SseEvent, SseStream}; pub use streaming_tool_call_argument_state::{ StreamingToolCallArgumentState, starts_with_structured_value, diff --git a/aimux-stream/src/ndjson.rs b/aimux-stream/src/ndjson.rs deleted file mode 100644 index ca6801a5..00000000 --- a/aimux-stream/src/ndjson.rs +++ /dev/null @@ -1,146 +0,0 @@ -//! NDJSON (newline-delimited JSON) stream decoder. - -use bytes::Bytes; -use futures::Stream; -use pin_project_lite::pin_project; -use serde::de::DeserializeOwned; -use std::pin::Pin; -use std::task::{Context, Poll}; -use thiserror::Error; - -/// Default upper bound on a single buffered NDJSON line's size (1 MiB). -const DEFAULT_MAX_LINE_SIZE: usize = 1024 * 1024; - -#[derive(Debug, Error)] -pub enum NdjsonError { - #[error("utf-8 error: {0}")] - Utf8(#[from] std::string::FromUtf8Error), - #[error("json error: {0}")] - Json(#[from] serde_json::Error), - #[error("stream error: {0}")] - Stream(String), - #[error("NDJSON line exceeded maximum allowed size")] - LineTooLarge, -} - -pin_project! { - /// Decodes a byte stream into parsed NDJSON values. - /// - /// Bytes are accumulated in a raw `Vec` buffer and split on `\n`. Each - /// complete line is strictly UTF-8 decoded only *after* reassembly, so a - /// multi-byte character split across two network chunks is never corrupted - /// into replacement chars — unlike a per-chunk - /// `String::from_utf8_lossy` decode, which would emit a `U+FFFD` on each - /// side of the split. - pub struct NdjsonStream { - #[pin] - inner: S, - buffer: Vec, - max_line_size: usize, - _marker: std::marker::PhantomData<(T, E)>, - } -} - -impl NdjsonStream -where - S: Stream> + Unpin, - T: DeserializeOwned, -{ - pub fn new(stream: S) -> Self { - Self::with_max_line_size(stream, DEFAULT_MAX_LINE_SIZE) - } - - /// Create an [`NdjsonStream`] with a custom per-line size limit. A line - /// longer than `max_line_size` bytes (excluding the `\n`) yields - /// [`NdjsonError::LineTooLarge`], as does a buffer that grows past the - /// limit while waiting for a newline. - pub fn with_max_line_size(stream: S, max_line_size: usize) -> Self { - Self { - inner: stream, - buffer: Vec::new(), - max_line_size, - _marker: std::marker::PhantomData, - } - } -} - -impl Stream for NdjsonStream -where - S: Stream> + Unpin, - T: DeserializeOwned, - E: std::fmt::Display, -{ - type Item = Result; - - fn poll_next(mut self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll> { - let this = self.as_mut().get_mut(); - - loop { - // Try to parse a complete JSON line. - if let Some(pos) = this.buffer.iter().position(|&b| b == b'\n') { - if pos > this.max_line_size { - // Drop the oversized line so a retried poll makes progress. - this.buffer.drain(..pos + 1); - return Poll::Ready(Some(Err(NdjsonError::LineTooLarge))); - } - // Extract the line bytes (without the newline) and decode - // strictly. Decoding the fully reassembled line (rather than - // `from_utf8_lossy` per chunk) preserves code points split - // across chunks and surfaces invalid UTF-8 as an error. - let line_bytes: Vec = this.buffer.drain(..pos).collect(); - this.buffer.drain(..1); // the newline - let line = match String::from_utf8(line_bytes) { - Ok(s) => s, - Err(e) => return Poll::Ready(Some(Err(NdjsonError::Utf8(e)))), - }; - let line = line.trim(); - if line.is_empty() { - continue; - } - match serde_json::from_str::(line) { - Ok(value) => return Poll::Ready(Some(Ok(value))), - Err(e) => return Poll::Ready(Some(Err(NdjsonError::Json(e)))), - } - } - - // No complete line yet. Guard against unbounded buffer growth when - // a newline never arrives. The partial can never form a valid line - // under the limit, so drop it — a retried poll then makes progress - // instead of looping on the same oversized buffer. - if this.buffer.len() > this.max_line_size { - this.buffer.clear(); - return Poll::Ready(Some(Err(NdjsonError::LineTooLarge))); - } - - // Read more bytes. - match Pin::new(&mut this.inner).poll_next(cx) { - Poll::Ready(Some(Ok(bytes))) => { - this.buffer.extend_from_slice(&bytes); - } - Poll::Ready(Some(Err(e))) => { - return Poll::Ready(Some(Err(NdjsonError::Stream(e.to_string())))); - } - Poll::Ready(None) => { - // Stream ended — try to parse any remaining buffer. - if this.buffer.len() > this.max_line_size { - return Poll::Ready(Some(Err(NdjsonError::LineTooLarge))); - } - let remaining_bytes = std::mem::take(&mut this.buffer); - let remaining = match String::from_utf8(remaining_bytes) { - Ok(s) => s, - Err(e) => return Poll::Ready(Some(Err(NdjsonError::Utf8(e)))), - }; - let remaining = remaining.trim(); - if remaining.is_empty() { - return Poll::Ready(None); - } - match serde_json::from_str::(remaining) { - Ok(value) => return Poll::Ready(Some(Ok(value))), - Err(e) => return Poll::Ready(Some(Err(NdjsonError::Json(e)))), - } - } - Poll::Pending => return Poll::Pending, - } - } - } -} diff --git a/aimux-stream/tests/ndjson_test.rs b/aimux-stream/tests/ndjson_test.rs deleted file mode 100644 index 88e75e1a..00000000 --- a/aimux-stream/tests/ndjson_test.rs +++ /dev/null @@ -1,98 +0,0 @@ -//! NDJSON stream decoder tests — cross-chunk UTF-8 reassembly, invalid-UTF-8 -//! rejection, and bounded-buffer behavior. - -use aimux_stream::{NdjsonError, NdjsonStream}; -use bytes::Bytes; -use futures::stream::{self, StreamExt}; -use serde::Deserialize; - -#[derive(Debug, Deserialize, PartialEq)] -struct Line { - text: String, -} - -/// Feed raw byte chunks into an [`NdjsonStream`] and collect the results. -async fn collect_lines(chunks: Vec>) -> Vec> { - let items: Vec> = - chunks.into_iter().map(|c| Ok(Bytes::from(c))).collect(); - let stream = NdjsonStream::new(stream::iter(items)); - stream.collect::>().await -} - -// ── cross-chunk UTF-8 reassembly (P0-03) ───────────────────────────────── - -#[tokio::test] -async fn emoji_ndjson_line_split_at_every_byte_boundary() { - // An NDJSON line whose value contains multi-byte emoji. Splitting the full - // line at *every* byte position must still reassemble to the intact text. - // The old per-chunk `String::from_utf8_lossy` decoder would corrupt a code - // point split across two chunks into two U+FFFDs. - let payload = "{\"text\":\"😀👋\"}\n"; - let bytes = payload.as_bytes(); - for split in 1..bytes.len() { - let (a, b) = bytes.split_at(split); - let lines = collect_lines(vec![a.to_vec(), b.to_vec()]).await; - assert_eq!( - lines.len(), - 1, - "split at byte {} produced {} results", - split, - lines.len() - ); - let line = lines[0] - .as_ref() - .unwrap_or_else(|e| panic!("split at byte {split} errored: {e:?}")); - assert_eq!( - line.text, "😀👋", - "split at byte {split} corrupted the text" - ); - } -} - -#[tokio::test] -async fn invalid_utf8_line_returns_utf8_error() { - // `{"text":"` followed by 0xF0 0x9F — the first two bytes of 😀 - // (U+1F600 = F0 9F 98 80) without the trailing bytes — is an incomplete - // 4-byte sequence. Strict decoding of the complete line surfaces a `Utf8` - // error instead of silently producing replacement chars. - let chunk = vec![ - b'{', b'"', b't', b'e', b'x', b't', b'"', b':', b'"', 0xF0, 0x9F, b'\n', - ]; - let lines = collect_lines(vec![chunk]).await; - assert_eq!(lines.len(), 1, "expected exactly one result"); - match &lines[0] { - Err(NdjsonError::Utf8(_)) => {} - other => panic!("expected NdjsonError::Utf8, got {other:?}"), - } -} - -// ── bounded buffers (P1-10) ───────────────────────────────────────────── - -#[tokio::test] -async fn oversized_line_returns_line_too_large() { - // A complete line longer than the 5-byte limit is rejected. - let stream = NdjsonStream::with_max_line_size( - stream::iter(vec![Ok::<_, std::io::Error>(Bytes::from_static( - b"{\"text\":\"hi\"}\n", - ))]), - 5, - ); - let results: Vec> = stream.collect().await; - assert_eq!(results.len(), 1); - assert!(matches!(results[0], Err(NdjsonError::LineTooLarge))); -} - -#[tokio::test] -async fn buffer_growing_past_limit_without_newline_returns_line_too_large() { - // No newline ever arrives, so the buffer would grow unboundedly; the limit - // trips instead (previously this allocated forever). - let stream = NdjsonStream::with_max_line_size( - stream::iter(vec![Ok::<_, std::io::Error>(Bytes::from_static( - b"{\"text\":\"no newline here\"", - ))]), - 5, - ); - let results: Vec> = stream.collect().await; - assert_eq!(results.len(), 1); - assert!(matches!(results[0], Err(NdjsonError::LineTooLarge))); -} diff --git a/docs/PROJECT-OVERVIEW.md b/docs/PROJECT-OVERVIEW.md index a5442aca..a53a672e 100644 --- a/docs/PROJECT-OVERVIEW.md +++ b/docs/PROJECT-OVERVIEW.md @@ -165,7 +165,7 @@ aimux/ │ ├── native protocols # standalone model + convert, handles provider-specific differences │ ├── OpenAI compatible # registry-backed: provider-registry.json + provider(name, ...) entry (RFC-0017 phase 4) │ └── modalities/search # voice / image / video / search implementations -├── aimux-stream # SSE / NDJSON streaming parsing +├── aimux-stream # SSE parsing, streamed tool-call tracking ├── aimux-provider-utils # One-exchange HTTP helpers, response handlers, API-key loading ├── aimux-ffi # C ABI (FFI infrastructure, shared by all bindings) └── bindings/ # 6 language bindings From eef51956373df49cc81fb4cf7e2abec05321c31a Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 29 Sep 2026 05:28:25 +0000 Subject: [PATCH 5/9] fix(stream): discard the rest of an SSE event after a UTF-8 error A line that failed UTF-8 decoding reset the event, but the lines after it (up to the blank line) were then dispatched as a separate fragment event, producing a second downstream error. Mark the event poisoned until the next blank line so one bad event yields exactly one SseError::Utf8. Also update the aimux-stream description in CONTRIBUTING. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_015EYDYDWYcPsDmjBjFeuDVe --- CHANGELOG.md | 4 ++-- CONTRIBUTING.md | 2 +- aimux-stream/src/sse.rs | 16 +++++++++++++--- aimux-stream/tests/sse_test.rs | 8 ++++---- 4 files changed, 20 insertions(+), 10 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 3cb55c7e..dce2b580 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -31,8 +31,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 `id` containing U+0000 is ignored, and `retry` must be all ASCII digits. Exceeding `max_event_size` (now: buffered event data plus the partial line) yields `SseError::FrameTooLarge` and ends the stream instead of skipping the - frame. Invalid UTF-8 still yields `SseError::Utf8`, now dropping only the - event being built. + frame. Invalid UTF-8 still yields one `SseError::Utf8` and discards that + event; later events are unaffected. - Removed `NdjsonStream` / `NdjsonError`: nothing in the workspace used them and the AI SDK has no counterpart. `tokio` is now a dev-dependency only and the unused direct `serde` dependency is dropped. diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index ce84ddbb..ab11942a 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -9,7 +9,7 @@ to set up a development environment, run the tests, and submit changes. aimux/ ├── aimux-core/ # Core abstractions: LanguageModel / Provider / Message / StreamPart ├── aimux-providers/ # provider implementations + cassettes (counts: docs/api/providers.md) -├── aimux-stream/ # SSE / NDJSON stream parsing +├── aimux-stream/ # SSE parsing, streamed tool-call tracking ├── aimux-provider-utils/ # HTTP utilities: retry, backoff, error parsing, API-key loading ├── aimux-ffi/ # C ABI (opaque handle + JSON + push callback) for non-native bindings ├── bindings/ # Node, Python, Swift, Kotlin, Flutter, Go, C — share one Rust core diff --git a/aimux-stream/src/sse.rs b/aimux-stream/src/sse.rs index 73eb360f..dacfc704 100644 --- a/aimux-stream/src/sse.rs +++ b/aimux-stream/src/sse.rs @@ -18,8 +18,8 @@ //! //! Two aimux additions: every line is strictly UTF-8 decoded after it has //! been reassembled (a code point split across chunks is never corrupted, and -//! invalid UTF-8 surfaces as [`SseError::Utf8`] and drops the event being -//! built), and buffered input is bounded by `max_event_size` (the parser's +//! invalid UTF-8 surfaces as [`SseError::Utf8`] and discards the rest of +//! that event up to the next blank line), and buffered input is bounded by `max_event_size` (the parser's //! `maxBufferSize`; exceeding it is fatal and ends the stream, as upstream). use std::collections::VecDeque; @@ -73,6 +73,9 @@ struct Parser { data_lines: usize, id: Option, retry: Option, + /// Set after a line fails UTF-8 decoding: the rest of the event (up to + /// the next blank line) is discarded rather than dispatched as a fragment. + poisoned: bool, max_size: usize, terminated: bool, ready: VecDeque>, @@ -89,6 +92,7 @@ impl Parser { data_lines: 0, id: None, retry: None, + poisoned: false, max_size, terminated: false, ready: VecDeque::new(), @@ -150,11 +154,16 @@ impl Parser { self.dispatch(); return; } + if self.poisoned { + return; + } let line = match String::from_utf8(line.to_vec()) { Ok(line) => line, Err(error) => { - // The event being built can no longer be trusted. + // The event being built can no longer be trusted: drop it and + // everything up to the next blank line. self.reset_event(); + self.poisoned = true; self.ready.push_back(Err(SseError::Utf8(error))); return; } @@ -205,6 +214,7 @@ impl Parser { } fn reset_event(&mut self) { + self.poisoned = false; self.event = None; self.data.clear(); self.data_lines = 0; diff --git a/aimux-stream/tests/sse_test.rs b/aimux-stream/tests/sse_test.rs index 58be9a19..80f46365 100644 --- a/aimux-stream/tests/sse_test.rs +++ b/aimux-stream/tests/sse_test.rs @@ -484,11 +484,11 @@ async fn invalid_utf8_drops_only_the_current_event() { bytes.extend_from_slice(&[0xE4, 0xBD]); bytes.extend_from_slice(b"\ndata: tail\n\ndata: next\n\n"); let events = collect_events_bytes(vec![bytes]).await; - assert_eq!(events.len(), 3); + // The rest of the poisoned event (`tail`) is discarded, not dispatched as + // a fragment; the following event is unaffected. + assert_eq!(events.len(), 2); assert!(matches!(events[0], Err(SseError::Utf8(_)))); - // The line after the bad one starts a fresh event buffer. - assert_eq!(data(events[1].as_ref().unwrap()), "tail"); - assert_eq!(data(events[2].as_ref().unwrap()), "next"); + assert_eq!(data(events[1].as_ref().unwrap()), "next"); } #[tokio::test] From 0644161c5280435f6b702b8a3c1b2ae4e1efd4b8 Mon Sep 17 00:00:00 2001 From: chenhaonan Date: Tue, 29 Sep 2026 14:06:23 +0800 Subject: [PATCH 6/9] fix(stream): collapse the SSE retry arm into a match guard clippy::collapsible_match rejects the nested `if` under the CI `-D warnings` gate (stable 1.98); behaviour is unchanged. Co-Authored-By: Claude Sonnet 5.5 --- aimux-stream/src/sse.rs | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/aimux-stream/src/sse.rs b/aimux-stream/src/sse.rs index dacfc704..7a5a4954 100644 --- a/aimux-stream/src/sse.rs +++ b/aimux-stream/src/sse.rs @@ -192,10 +192,8 @@ impl Parser { self.id = Some(value.to_string()); } } - "retry" => { - if !value.is_empty() && value.bytes().all(|b| b.is_ascii_digit()) { - self.retry = value.parse().ok(); - } + "retry" if !value.is_empty() && value.bytes().all(|b| b.is_ascii_digit()) => { + self.retry = value.parse().ok(); } _ => {} // unknown field } From d47746f2d747f3fad8fb9faba5c29e01e33b5767 Mon Sep 17 00:00:00 2001 From: chenhaonan Date: Tue, 29 Sep 2026 22:33:35 +0800 Subject: [PATCH 7/9] test(stream): replace hand-written SSE tests with eventsource-parser's suite sse_test.rs was a behaviour list written from scratch, and its header claimed the AI SDK has no SSE parser tests. The parser behind parseJsonEventStream, eventsource-parser, ships its own suite. Port test/parse.test.ts and test/stream.test.ts (v3.1.1, MIT) case for case, driven by the same fixtures; test/multibyte.ts is checked in as JSON and read with include_str!. Cases with no Rust counterpart (onComment counts, reset(), onError payloads) are listed in the file header, as are the adaptations (retry is reported on the event, maxBufferSize maps to with_max_event_size). Four aimux-only tests remain: strict UTF-8, byte-level chunk splitting, transport errors, and the default limit. The four unit tests inside sse.rs are dropped; the port covers them. sha2 (already used by aimux-providers) is a dev-dependency for the upstream SHA-256 check on the 4.8M-character event. The tracker and argument-state tests were already case-for-case ports of the pinned upstream tests; only a comment is added. Co-Authored-By: Claude Sonnet 5.5 --- aimux-stream/Cargo.toml | 1 + aimux-stream/src/sse.rs | 36 - .../eventsource_parser_multibyte.json | 1 + aimux-stream/tests/sse_test.rs | 1117 +++++++++++------ ...streaming_tool_call_argument_state_test.rs | 1 + 5 files changed, 724 insertions(+), 432 deletions(-) create mode 100644 aimux-stream/tests/fixtures/eventsource_parser_multibyte.json diff --git a/aimux-stream/Cargo.toml b/aimux-stream/Cargo.toml index 7c65cfec..2d51d4d6 100644 --- a/aimux-stream/Cargo.toml +++ b/aimux-stream/Cargo.toml @@ -16,6 +16,7 @@ serde_json = { workspace = true } thiserror = { workspace = true } [dev-dependencies] +sha2 = "0.10" tokio = { workspace = true } [lints] diff --git a/aimux-stream/src/sse.rs b/aimux-stream/src/sse.rs index 7a5a4954..a9be7a66 100644 --- a/aimux-stream/src/sse.rs +++ b/aimux-stream/src/sse.rs @@ -295,39 +295,3 @@ where } } } - -#[cfg(test)] -mod tests { - use super::*; - - fn parse(input: &str) -> Vec { - let mut parser = Parser::new(DEFAULT_MAX_EVENT_SIZE); - parser.feed(input.as_bytes()); - parser.ready.into_iter().map(Result::unwrap).collect() - } - - #[test] - fn parse_single_event() { - assert_eq!(parse("data: hello world\n\n")[0].data, "hello world"); - } - - #[test] - fn parse_multi_line_data() { - assert_eq!( - parse("data: line1\ndata: line2\n\n")[0].data, - "line1\nline2" - ); - } - - #[test] - fn parse_event_with_type() { - let events = parse("event: message\ndata: payload\n\n"); - assert_eq!(events[0].event.as_deref(), Some("message")); - assert_eq!(events[0].data, "payload"); - } - - #[test] - fn parse_event_without_data_is_not_dispatched() { - assert!(parse("event: ping\n\n").is_empty()); - } -} diff --git a/aimux-stream/tests/fixtures/eventsource_parser_multibyte.json b/aimux-stream/tests/fixtures/eventsource_parser_multibyte.json new file mode 100644 index 00000000..f427553c --- /dev/null +++ b/aimux-stream/tests/fixtures/eventsource_parser_multibyte.json @@ -0,0 +1 @@ +{"lines":["Blåbærsyltetøy må da være mulig å oppdrive?","በእርግጥ አንድ ሰው ብሉቤሪ ጃምን መግዛት መቻል አለበት?","بالتأكيد يجب أن يكون المرء قادرا على شراء مربى التوت؟","一定能买到蓝莓果酱吗?","ચોક્કસ એક બ્લુબેરી જામ મેળવવા માટે સમર્થ હોવા જ જોઈએ?","בטח אפשר להשיג ריבת אוכמניות?","निश्चित रूप से किसी को ब्लूबेरी जैम खरीदने में सक्षम होना चाहिए?","Bê guman pêdivî ye ku meriv karibe jama şînê peyda bike?","ប្រាកដណាស់ មនុស្សម្នាក់ត្រូវតែអាចទិញយៈសាពូនមី blueberry បានទេ?","ಖಂಡಿತವಾಗಿಯೂ ಒಬ್ಬರು ಬ್ಲೂಬೆರ್ರಿ ಜಾಮ್ ಅನ್ನು ಸಂಗ್ರಹಿಸಲು ಶಕ್ತರಾಗಿರಬೇಕು?","確かにブルーベリージャムを調達できなければなりませんか?","Невже треба вміти добути варення з чорниці?","แน่นอนว่าจะต้องสามารถจัดหาแยมบลูเบอร์รี่ได้?","Сигурно некој мора да може да набави џем од боровинки?"],"emojis":["😶‍🌫️","😮‍💨","😵‍💫","❤️‍🔥","❤️‍🩹","👁️‍🗨️","🫱🏻‍🫲🏼","🫱🏻‍🫲🏽","🫱🏻‍🫲🏾","🫱🏻‍🫲🏿","🫱🏼‍🫲🏻","🫱🏼‍🫲🏽","🫱🏼‍🫲🏾","🫱🏼‍🫲🏿","🫱🏽‍🫲🏻","🫱🏽‍🫲🏼","🫱🏽‍🫲🏾","🫱🏽‍🫲🏿","🫱🏾‍🫲🏻","🫱🏾‍🫲🏼","🫱🏾‍🫲🏽","🫱🏾‍🫲🏿","🫱🏿‍🫲🏻","🫱🏿‍🫲🏼","🫱🏿‍🫲🏽","🫱🏿‍🫲🏾","🧔‍♂️","🧔🏻‍♂️","🧔🏼‍♂️","🧔🏽‍♂️","🧔🏾‍♂️","🧔🏿‍♂️","🧔‍♀️","🧔🏻‍♀️","🧔🏼‍♀️","🧔🏽‍♀️","🧔🏾‍♀️","🧔🏿‍♀️","👨‍🦰","👨🏻‍🦰","👨🏼‍🦰","👨🏽‍🦰","👨🏾‍🦰","👨🏿‍🦰","👨‍🦱","👨🏻‍🦱","👨🏼‍🦱","👨🏽‍🦱","👨🏾‍🦱","👨🏿‍🦱","👨‍🦳","👨🏻‍🦳","👨🏼‍🦳","👨🏽‍🦳","👨🏾‍🦳","👨🏿‍🦳","👨‍🦲","👨🏻‍🦲","👨🏼‍🦲","👨🏽‍🦲","👨🏾‍🦲","👨🏿‍🦲","👩‍🦰","👩🏻‍🦰","👩🏼‍🦰","👩🏽‍🦰","👩🏾‍🦰","👩🏿‍🦰","🧑‍🦰","🧑🏻‍🦰","🧑🏼‍🦰","🧑🏽‍🦰","🧑🏾‍🦰","🧑🏿‍🦰","👩‍🦱","👩🏻‍🦱","👩🏼‍🦱","👩🏽‍🦱","👩🏾‍🦱","👩🏿‍🦱","🧑‍🦱","🧑🏻‍🦱","🧑🏼‍🦱","🧑🏽‍🦱","🧑🏾‍🦱","🧑🏿‍🦱","👩‍🦳","👩🏻‍🦳","👩🏼‍🦳","👩🏽‍🦳","👩🏾‍🦳","👩🏿‍🦳","🧑‍🦳","🧑🏻‍🦳","🧑🏼‍🦳","🧑🏽‍🦳","🧑🏾‍🦳","🧑🏿‍🦳","👩‍🦲","👩🏻‍🦲","👩🏼‍🦲","👩🏽‍🦲","👩🏾‍🦲","👩🏿‍🦲","🧑‍🦲","🧑🏻‍🦲","🧑🏼‍🦲","🧑🏽‍🦲","🧑🏾‍🦲","🧑🏿‍🦲","👱‍♀️","👱🏻‍♀️","👱🏼‍♀️","👱🏽‍♀️","👱🏾‍♀️","👱🏿‍♀️","👱‍♂️","👱🏻‍♂️","👱🏼‍♂️","👱🏽‍♂️","👱🏾‍♂️","👱🏿‍♂️","🙍‍♂️","🙍🏻‍♂️","🙍🏼‍♂️","🙍🏽‍♂️","🙍🏾‍♂️","🙍🏿‍♂️","🙍‍♀️","🙍🏻‍♀️","🙍🏼‍♀️","🙍🏽‍♀️","🙍🏾‍♀️","🙍🏿‍♀️","🙎‍♂️","🙎🏻‍♂️","🙎🏼‍♂️","🙎🏽‍♂️","🙎🏾‍♂️","🙎🏿‍♂️","🙎‍♀️","🙎🏻‍♀️","🙎🏼‍♀️","🙎🏽‍♀️","🙎🏾‍♀️","🙎🏿‍♀️","🙅‍♂️","🙅🏻‍♂️","🙅🏼‍♂️","🙅🏽‍♂️","🙅🏾‍♂️","🙅🏿‍♂️","🙅‍♀️","🙅🏻‍♀️","🙅🏼‍♀️","🙅🏽‍♀️","🙅🏾‍♀️","🙅🏿‍♀️","🙆‍♂️","🙆🏻‍♂️","🙆🏼‍♂️","🙆🏽‍♂️","🙆🏾‍♂️","🙆🏿‍♂️","🙆‍♀️","🙆🏻‍♀️","🙆🏼‍♀️","🙆🏽‍♀️","🙆🏾‍♀️","🙆🏿‍♀️","💁‍♂️","💁🏻‍♂️","💁🏼‍♂️","💁🏽‍♂️","💁🏾‍♂️","💁🏿‍♂️","💁‍♀️","💁🏻‍♀️","💁🏼‍♀️","💁🏽‍♀️","💁🏾‍♀️","💁🏿‍♀️","🙋‍♂️","🙋🏻‍♂️","🙋🏼‍♂️","🙋🏽‍♂️","🙋🏾‍♂️","🙋🏿‍♂️","🙋‍♀️","🙋🏻‍♀️","🙋🏼‍♀️","🙋🏽‍♀️","🙋🏾‍♀️","🙋🏿‍♀️","🧏‍♂️","🧏🏻‍♂️","🧏🏼‍♂️","🧏🏽‍♂️","🧏🏾‍♂️","🧏🏿‍♂️","🧏‍♀️","🧏🏻‍♀️","🧏🏼‍♀️","🧏🏽‍♀️","🧏🏾‍♀️","🧏🏿‍♀️","🙇‍♂️","🙇🏻‍♂️","🙇🏼‍♂️","🙇🏽‍♂️","🙇🏾‍♂️","🙇🏿‍♂️","🙇‍♀️","🙇🏻‍♀️","🙇🏼‍♀️","🙇🏽‍♀️","🙇🏾‍♀️","🙇🏿‍♀️","🤦‍♂️","🤦🏻‍♂️","🤦🏼‍♂️","🤦🏽‍♂️","🤦🏾‍♂️","🤦🏿‍♂️","🤦‍♀️","🤦🏻‍♀️","🤦🏼‍♀️","🤦🏽‍♀️","🤦🏾‍♀️","🤦🏿‍♀️","🤷‍♂️","🤷🏻‍♂️","🤷🏼‍♂️","🤷🏽‍♂️","🤷🏾‍♂️","🤷🏿‍♂️","🤷‍♀️","🤷🏻‍♀️","🤷🏼‍♀️","🤷🏽‍♀️","🤷🏾‍♀️","🤷🏿‍♀️","🧑‍⚕️","🧑🏻‍⚕️","🧑🏼‍⚕️","🧑🏽‍⚕️","🧑🏾‍⚕️","🧑🏿‍⚕️","👨‍⚕️","👨🏻‍⚕️","👨🏼‍⚕️","👨🏽‍⚕️","👨🏾‍⚕️","👨🏿‍⚕️","👩‍⚕️","👩🏻‍⚕️","👩🏼‍⚕️","👩🏽‍⚕️","👩🏾‍⚕️","👩🏿‍⚕️","🧑‍🎓","🧑🏻‍🎓","🧑🏼‍🎓","🧑🏽‍🎓","🧑🏾‍🎓","🧑🏿‍🎓","👨‍🎓","👨🏻‍🎓","👨🏼‍🎓","👨🏽‍🎓","👨🏾‍🎓","👨🏿‍🎓","👩‍🎓","👩🏻‍🎓","👩🏼‍🎓","👩🏽‍🎓","👩🏾‍🎓","👩🏿‍🎓","🧑‍🏫","🧑🏻‍🏫","🧑🏼‍🏫","🧑🏽‍🏫","🧑🏾‍🏫","🧑🏿‍🏫","👨‍🏫","👨🏻‍🏫","👨🏼‍🏫","👨🏽‍🏫","👨🏾‍🏫","👨🏿‍🏫","👩‍🏫","👩🏻‍🏫","👩🏼‍🏫","👩🏽‍🏫","👩🏾‍🏫","👩🏿‍🏫","🧑‍⚖️","🧑🏻‍⚖️","🧑🏼‍⚖️","🧑🏽‍⚖️","🧑🏾‍⚖️","🧑🏿‍⚖️","👨‍⚖️","👨🏻‍⚖️","👨🏼‍⚖️","👨🏽‍⚖️","👨🏾‍⚖️","👨🏿‍⚖️","👩‍⚖️","👩🏻‍⚖️","👩🏼‍⚖️","👩🏽‍⚖️","👩🏾‍⚖️","👩🏿‍⚖️","🧑‍🌾","🧑🏻‍🌾","🧑🏼‍🌾","🧑🏽‍🌾","🧑🏾‍🌾","🧑🏿‍🌾","👨‍🌾","👨🏻‍🌾","👨🏼‍🌾","👨🏽‍🌾","👨🏾‍🌾","👨🏿‍🌾","👩‍🌾","👩🏻‍🌾","👩🏼‍🌾","👩🏽‍🌾","👩🏾‍🌾","👩🏿‍🌾","🧑‍🍳","🧑🏻‍🍳","🧑🏼‍🍳","🧑🏽‍🍳","🧑🏾‍🍳","🧑🏿‍🍳","👨‍🍳","👨🏻‍🍳","👨🏼‍🍳","👨🏽‍🍳","👨🏾‍🍳","👨🏿‍🍳","👩‍🍳","👩🏻‍🍳","👩🏼‍🍳","👩🏽‍🍳","👩🏾‍🍳","👩🏿‍🍳","🧑‍🔧","🧑🏻‍🔧","🧑🏼‍🔧","🧑🏽‍🔧","🧑🏾‍🔧","🧑🏿‍🔧","👨‍🔧","👨🏻‍🔧","👨🏼‍🔧","👨🏽‍🔧","👨🏾‍🔧","👨🏿‍🔧","👩‍🔧","👩🏻‍🔧","👩🏼‍🔧","👩🏽‍🔧","👩🏾‍🔧","👩🏿‍🔧","🧑‍🏭","🧑🏻‍🏭","🧑🏼‍🏭","🧑🏽‍🏭","🧑🏾‍🏭","🧑🏿‍🏭","👨‍🏭","👨🏻‍🏭","👨🏼‍🏭","👨🏽‍🏭","👨🏾‍🏭","👨🏿‍🏭","👩‍🏭","👩🏻‍🏭","👩🏼‍🏭","👩🏽‍🏭","👩🏾‍🏭","👩🏿‍🏭","🧑‍💼","🧑🏻‍💼","🧑🏼‍💼","🧑🏽‍💼","🧑🏾‍💼","🧑🏿‍💼","👨‍💼","👨🏻‍💼","👨🏼‍💼","👨🏽‍💼","👨🏾‍💼","👨🏿‍💼","👩‍💼","👩🏻‍💼","👩🏼‍💼","👩🏽‍💼","👩🏾‍💼","👩🏿‍💼","🧑‍🔬","🧑🏻‍🔬","🧑🏼‍🔬","🧑🏽‍🔬","🧑🏾‍🔬","🧑🏿‍🔬","👨‍🔬","👨🏻‍🔬","👨🏼‍🔬","👨🏽‍🔬","👨🏾‍🔬","👨🏿‍🔬","👩‍🔬","👩🏻‍🔬","👩🏼‍🔬","👩🏽‍🔬","👩🏾‍🔬","👩🏿‍🔬","🧑‍💻","🧑🏻‍💻","🧑🏼‍💻","🧑🏽‍💻","🧑🏾‍💻","🧑🏿‍💻","👨‍💻","👨🏻‍💻","👨🏼‍💻","👨🏽‍💻","👨🏾‍💻","👨🏿‍💻","👩‍💻","👩🏻‍💻","👩🏼‍💻","👩🏽‍💻","👩🏾‍💻","👩🏿‍💻","🧑‍🎤","🧑🏻‍🎤","🧑🏼‍🎤","🧑🏽‍🎤","🧑🏾‍🎤","🧑🏿‍🎤","👨‍🎤","👨🏻‍🎤","👨🏼‍🎤","👨🏽‍🎤","👨🏾‍🎤","👨🏿‍🎤","👩‍🎤","👩🏻‍🎤","👩🏼‍🎤","👩🏽‍🎤","👩🏾‍🎤","👩🏿‍🎤","🧑‍🎨","🧑🏻‍🎨","🧑🏼‍🎨","🧑🏽‍🎨","🧑🏾‍🎨","🧑🏿‍🎨","👨‍🎨","👨🏻‍🎨","👨🏼‍🎨","👨🏽‍🎨","👨🏾‍🎨","👨🏿‍🎨","👩‍🎨","👩🏻‍🎨","👩🏼‍🎨","👩🏽‍🎨","👩🏾‍🎨","👩🏿‍🎨","🧑‍✈️","🧑🏻‍✈️","🧑🏼‍✈️","🧑🏽‍✈️","🧑🏾‍✈️","🧑🏿‍✈️","👨‍✈️","👨🏻‍✈️","👨🏼‍✈️","👨🏽‍✈️","👨🏾‍✈️","👨🏿‍✈️","👩‍✈️","👩🏻‍✈️","👩🏼‍✈️","👩🏽‍✈️","👩🏾‍✈️","👩🏿‍✈️","🧑‍🚀","🧑🏻‍🚀","🧑🏼‍🚀","🧑🏽‍🚀","🧑🏾‍🚀","🧑🏿‍🚀","👨‍🚀","👨🏻‍🚀","👨🏼‍🚀","👨🏽‍🚀","👨🏾‍🚀","👨🏿‍🚀","👩‍🚀","👩🏻‍🚀","👩🏼‍🚀","👩🏽‍🚀","👩🏾‍🚀","👩🏿‍🚀","🧑‍🚒","🧑🏻‍🚒","🧑🏼‍🚒","🧑🏽‍🚒","🧑🏾‍🚒","🧑🏿‍🚒","👨‍🚒","👨🏻‍🚒","👨🏼‍🚒","👨🏽‍🚒","👨🏾‍🚒","👨🏿‍🚒","👩‍🚒","👩🏻‍🚒","👩🏼‍🚒","👩🏽‍🚒","👩🏾‍🚒","👩🏿‍🚒","👮‍♂️","👮🏻‍♂️","👮🏼‍♂️","👮🏽‍♂️","👮🏾‍♂️","👮🏿‍♂️","👮‍♀️","👮🏻‍♀️","👮🏼‍♀️","👮🏽‍♀️","👮🏾‍♀️","👮🏿‍♀️","🕵️‍♂️","🕵🏻‍♂️","🕵🏼‍♂️","🕵🏽‍♂️","🕵🏾‍♂️","🕵🏿‍♂️","🕵️‍♀️","🕵🏻‍♀️","🕵🏼‍♀️","🕵🏽‍♀️","🕵🏾‍♀️","🕵🏿‍♀️","💂‍♂️","💂🏻‍♂️","💂🏼‍♂️","💂🏽‍♂️","💂🏾‍♂️","💂🏿‍♂️","💂‍♀️","💂🏻‍♀️","💂🏼‍♀️","💂🏽‍♀️","💂🏾‍♀️","💂🏿‍♀️","👷‍♂️","👷🏻‍♂️","👷🏼‍♂️","👷🏽‍♂️","👷🏾‍♂️","👷🏿‍♂️","👷‍♀️","👷🏻‍♀️","👷🏼‍♀️","👷🏽‍♀️","👷🏾‍♀️","👷🏿‍♀️","👳‍♂️","👳🏻‍♂️","👳🏼‍♂️","👳🏽‍♂️","👳🏾‍♂️","👳🏿‍♂️","👳‍♀️","👳🏻‍♀️","👳🏼‍♀️","👳🏽‍♀️","👳🏾‍♀️","👳🏿‍♀️","🤵‍♂️","🤵🏻‍♂️","🤵🏼‍♂️","🤵🏽‍♂️","🤵🏾‍♂️","🤵🏿‍♂️","🤵‍♀️","🤵🏻‍♀️","🤵🏼‍♀️","🤵🏽‍♀️","🤵🏾‍♀️","🤵🏿‍♀️","👰‍♂️","👰🏻‍♂️","👰🏼‍♂️","👰🏽‍♂️","👰🏾‍♂️","👰🏿‍♂️","👰‍♀️","👰🏻‍♀️","👰🏼‍♀️","👰🏽‍♀️","👰🏾‍♀️","👰🏿‍♀️","👩‍🍼","👩🏻‍🍼","👩🏼‍🍼","👩🏽‍🍼","👩🏾‍🍼","👩🏿‍🍼","👨‍🍼","👨🏻‍🍼","👨🏼‍🍼","👨🏽‍🍼","👨🏾‍🍼","👨🏿‍🍼","🧑‍🍼","🧑🏻‍🍼","🧑🏼‍🍼","🧑🏽‍🍼","🧑🏾‍🍼","🧑🏿‍🍼","🧑‍🎄","🧑🏻‍🎄","🧑🏼‍🎄","🧑🏽‍🎄","🧑🏾‍🎄","🧑🏿‍🎄","🦸‍♂️","🦸🏻‍♂️","🦸🏼‍♂️","🦸🏽‍♂️","🦸🏾‍♂️","🦸🏿‍♂️","🦸‍♀️","🦸🏻‍♀️","🦸🏼‍♀️","🦸🏽‍♀️","🦸🏾‍♀️","🦸🏿‍♀️","🦹‍♂️","🦹🏻‍♂️","🦹🏼‍♂️","🦹🏽‍♂️","🦹🏾‍♂️","🦹🏿‍♂️","🦹‍♀️","🦹🏻‍♀️","🦹🏼‍♀️","🦹🏽‍♀️","🦹🏾‍♀️","🦹🏿‍♀️","🧙‍♂️","🧙🏻‍♂️","🧙🏼‍♂️","🧙🏽‍♂️","🧙🏾‍♂️","🧙🏿‍♂️","🧙‍♀️","🧙🏻‍♀️","🧙🏼‍♀️","🧙🏽‍♀️","🧙🏾‍♀️","🧙🏿‍♀️","🧚‍♂️","🧚🏻‍♂️","🧚🏼‍♂️","🧚🏽‍♂️","🧚🏾‍♂️","🧚🏿‍♂️","🧚‍♀️","🧚🏻‍♀️","🧚🏼‍♀️","🧚🏽‍♀️","🧚🏾‍♀️","🧚🏿‍♀️","🧛‍♂️","🧛🏻‍♂️","🧛🏼‍♂️","🧛🏽‍♂️","🧛🏾‍♂️","🧛🏿‍♂️","🧛‍♀️","🧛🏻‍♀️","🧛🏼‍♀️","🧛🏽‍♀️","🧛🏾‍♀️","🧛🏿‍♀️","🧜‍♂️","🧜🏻‍♂️","🧜🏼‍♂️","🧜🏽‍♂️","🧜🏾‍♂️","🧜🏿‍♂️","🧜‍♀️","🧜🏻‍♀️","🧜🏼‍♀️","🧜🏽‍♀️","🧜🏾‍♀️","🧜🏿‍♀️","🧝‍♂️","🧝🏻‍♂️","🧝🏼‍♂️","🧝🏽‍♂️","🧝🏾‍♂️","🧝🏿‍♂️","🧝‍♀️","🧝🏻‍♀️","🧝🏼‍♀️","🧝🏽‍♀️","🧝🏾‍♀️","🧝🏿‍♀️","🧞‍♂️","🧞‍♀️","🧟‍♂️","🧟‍♀️","💆‍♂️","💆🏻‍♂️","💆🏼‍♂️","💆🏽‍♂️","💆🏾‍♂️","💆🏿‍♂️","💆‍♀️","💆🏻‍♀️","💆🏼‍♀️","💆🏽‍♀️","💆🏾‍♀️","💆🏿‍♀️","💇‍♂️","💇🏻‍♂️","💇🏼‍♂️","💇🏽‍♂️","💇🏾‍♂️","💇🏿‍♂️","💇‍♀️","💇🏻‍♀️","💇🏼‍♀️","💇🏽‍♀️","💇🏾‍♀️","💇🏿‍♀️","🚶‍♂️","🚶🏻‍♂️","🚶🏼‍♂️","🚶🏽‍♂️","🚶🏾‍♂️","🚶🏿‍♂️","🚶‍♀️","🚶🏻‍♀️","🚶🏼‍♀️","🚶🏽‍♀️","🚶🏾‍♀️","🚶🏿‍♀️","🧍‍♂️","🧍🏻‍♂️","🧍🏼‍♂️","🧍🏽‍♂️","🧍🏾‍♂️","🧍🏿‍♂️","🧍‍♀️","🧍🏻‍♀️","🧍🏼‍♀️","🧍🏽‍♀️","🧍🏾‍♀️","🧍🏿‍♀️","🧎‍♂️","🧎🏻‍♂️","🧎🏼‍♂️","🧎🏽‍♂️","🧎🏾‍♂️","🧎🏿‍♂️","🧎‍♀️","🧎🏻‍♀️","🧎🏼‍♀️","🧎🏽‍♀️","🧎🏾‍♀️","🧎🏿‍♀️","🧑‍🦯","🧑🏻‍🦯","🧑🏼‍🦯","🧑🏽‍🦯","🧑🏾‍🦯","🧑🏿‍🦯","👨‍🦯","👨🏻‍🦯","👨🏼‍🦯","👨🏽‍🦯","👨🏾‍🦯","👨🏿‍🦯","👩‍🦯","👩🏻‍🦯","👩🏼‍🦯","👩🏽‍🦯","👩🏾‍🦯","👩🏿‍🦯","🧑‍🦼","🧑🏻‍🦼","🧑🏼‍🦼","🧑🏽‍🦼","🧑🏾‍🦼","🧑🏿‍🦼","👨‍🦼","👨🏻‍🦼","👨🏼‍🦼","👨🏽‍🦼","👨🏾‍🦼","👨🏿‍🦼","👩‍🦼","👩🏻‍🦼","👩🏼‍🦼","👩🏽‍🦼","👩🏾‍🦼","👩🏿‍🦼","🧑‍🦽","🧑🏻‍🦽","🧑🏼‍🦽","🧑🏽‍🦽","🧑🏾‍🦽","🧑🏿‍🦽","👨‍🦽","👨🏻‍🦽","👨🏼‍🦽","👨🏽‍🦽","👨🏾‍🦽","👨🏿‍🦽","👩‍🦽","👩🏻‍🦽","👩🏼‍🦽","👩🏽‍🦽","👩🏾‍🦽","👩🏿‍🦽","🏃‍♂️","🏃🏻‍♂️","🏃🏼‍♂️","🏃🏽‍♂️","🏃🏾‍♂️","🏃🏿‍♂️","🏃‍♀️","🏃🏻‍♀️","🏃🏼‍♀️","🏃🏽‍♀️","🏃🏾‍♀️","🏃🏿‍♀️","👯‍♂️","👯‍♀️","🧖‍♂️","🧖🏻‍♂️","🧖🏼‍♂️","🧖🏽‍♂️","🧖🏾‍♂️","🧖🏿‍♂️","🧖‍♀️","🧖🏻‍♀️","🧖🏼‍♀️","🧖🏽‍♀️","🧖🏾‍♀️","🧖🏿‍♀️","🧗‍♂️","🧗🏻‍♂️","🧗🏼‍♂️","🧗🏽‍♂️","🧗🏾‍♂️","🧗🏿‍♂️","🧗‍♀️","🧗🏻‍♀️","🧗🏼‍♀️","🧗🏽‍♀️","🧗🏾‍♀️","🧗🏿‍♀️","🏌️‍♂️","🏌🏻‍♂️","🏌🏼‍♂️","🏌🏽‍♂️","🏌🏾‍♂️","🏌🏿‍♂️","🏌️‍♀️","🏌🏻‍♀️","🏌🏼‍♀️","🏌🏽‍♀️","🏌🏾‍♀️","🏌🏿‍♀️","🏄‍♂️","🏄🏻‍♂️","🏄🏼‍♂️","🏄🏽‍♂️","🏄🏾‍♂️","🏄🏿‍♂️","🏄‍♀️","🏄🏻‍♀️","🏄🏼‍♀️","🏄🏽‍♀️","🏄🏾‍♀️","🏄🏿‍♀️","🚣‍♂️","🚣🏻‍♂️","🚣🏼‍♂️","🚣🏽‍♂️","🚣🏾‍♂️","🚣🏿‍♂️","🚣‍♀️","🚣🏻‍♀️","🚣🏼‍♀️","🚣🏽‍♀️","🚣🏾‍♀️","🚣🏿‍♀️","🏊‍♂️","🏊🏻‍♂️","🏊🏼‍♂️","🏊🏽‍♂️","🏊🏾‍♂️","🏊🏿‍♂️","🏊‍♀️","🏊🏻‍♀️","🏊🏼‍♀️","🏊🏽‍♀️","🏊🏾‍♀️","🏊🏿‍♀️","⛹️‍♂️","⛹🏻‍♂️","⛹🏼‍♂️","⛹🏽‍♂️","⛹🏾‍♂️","⛹🏿‍♂️","⛹️‍♀️","⛹🏻‍♀️","⛹🏼‍♀️","⛹🏽‍♀️","⛹🏾‍♀️","⛹🏿‍♀️","🏋️‍♂️","🏋🏻‍♂️","🏋🏼‍♂️","🏋🏽‍♂️","🏋🏾‍♂️","🏋🏿‍♂️","🏋️‍♀️","🏋🏻‍♀️","🏋🏼‍♀️","🏋🏽‍♀️","🏋🏾‍♀️","🏋🏿‍♀️","🚴‍♂️","🚴🏻‍♂️","🚴🏼‍♂️","🚴🏽‍♂️","🚴🏾‍♂️","🚴🏿‍♂️","🚴‍♀️","🚴🏻‍♀️","🚴🏼‍♀️","🚴🏽‍♀️","🚴🏾‍♀️","🚴🏿‍♀️","🚵‍♂️","🚵🏻‍♂️","🚵🏼‍♂️","🚵🏽‍♂️","🚵🏾‍♂️","🚵🏿‍♂️","🚵‍♀️","🚵🏻‍♀️","🚵🏼‍♀️","🚵🏽‍♀️","🚵🏾‍♀️","🚵🏿‍♀️","🤸‍♂️","🤸🏻‍♂️","🤸🏼‍♂️","🤸🏽‍♂️","🤸🏾‍♂️","🤸🏿‍♂️","🤸‍♀️","🤸🏻‍♀️","🤸🏼‍♀️","🤸🏽‍♀️","🤸🏾‍♀️","🤸🏿‍♀️","🤼‍♂️","🤼‍♀️","🤽‍♂️","🤽🏻‍♂️","🤽🏼‍♂️","🤽🏽‍♂️","🤽🏾‍♂️","🤽🏿‍♂️","🤽‍♀️","🤽🏻‍♀️","🤽🏼‍♀️","🤽🏽‍♀️","🤽🏾‍♀️","🤽🏿‍♀️","🤾‍♂️","🤾🏻‍♂️","🤾🏼‍♂️","🤾🏽‍♂️","🤾🏾‍♂️","🤾🏿‍♂️","🤾‍♀️","🤾🏻‍♀️","🤾🏼‍♀️","🤾🏽‍♀️","🤾🏾‍♀️","🤾🏿‍♀️","🤹‍♂️","🤹🏻‍♂️","🤹🏼‍♂️","🤹🏽‍♂️","🤹🏾‍♂️","🤹🏿‍♂️","🤹‍♀️","🤹🏻‍♀️","🤹🏼‍♀️","🤹🏽‍♀️","🤹🏾‍♀️","🤹🏿‍♀️","🧘‍♂️","🧘🏻‍♂️","🧘🏼‍♂️","🧘🏽‍♂️","🧘🏾‍♂️","🧘🏿‍♂️","🧘‍♀️","🧘🏻‍♀️","🧘🏼‍♀️","🧘🏽‍♀️","🧘🏾‍♀️","🧘🏿‍♀️","🧑‍🤝‍🧑","🧑🏻‍🤝‍🧑🏻","🧑🏻‍🤝‍🧑🏼","🧑🏻‍🤝‍🧑🏽","🧑🏻‍🤝‍🧑🏾","🧑🏻‍🤝‍🧑🏿","🧑🏼‍🤝‍🧑🏻","🧑🏼‍🤝‍🧑🏼","🧑🏼‍🤝‍🧑🏽","🧑🏼‍🤝‍🧑🏾","🧑🏼‍🤝‍🧑🏿","🧑🏽‍🤝‍🧑🏻","🧑🏽‍🤝‍🧑🏼","🧑🏽‍🤝‍🧑🏽","🧑🏽‍🤝‍🧑🏾","🧑🏽‍🤝‍🧑🏿","🧑🏾‍🤝‍🧑🏻","🧑🏾‍🤝‍🧑🏼","🧑🏾‍🤝‍🧑🏽","🧑🏾‍🤝‍🧑🏾","🧑🏾‍🤝‍🧑🏿","🧑🏿‍🤝‍🧑🏻","🧑🏿‍🤝‍🧑🏼","🧑🏿‍🤝‍🧑🏽","🧑🏿‍🤝‍🧑🏾","🧑🏿‍🤝‍🧑🏿","👩🏻‍🤝‍👩🏼","👩🏻‍🤝‍👩🏽","👩🏻‍🤝‍👩🏾","👩🏻‍🤝‍👩🏿","👩🏼‍🤝‍👩🏻","👩🏼‍🤝‍👩🏽","👩🏼‍🤝‍👩🏾","👩🏼‍🤝‍👩🏿","👩🏽‍🤝‍👩🏻","👩🏽‍🤝‍👩🏼","👩🏽‍🤝‍👩🏾","👩🏽‍🤝‍👩🏿","👩🏾‍🤝‍👩🏻","👩🏾‍🤝‍👩🏼","👩🏾‍🤝‍👩🏽","👩🏾‍🤝‍👩🏿","👩🏿‍🤝‍👩🏻","👩🏿‍🤝‍👩🏼","👩🏿‍🤝‍👩🏽","👩🏿‍🤝‍👩🏾","👩🏻‍🤝‍👨🏼","👩🏻‍🤝‍👨🏽","👩🏻‍🤝‍👨🏾","👩🏻‍🤝‍👨🏿","👩🏼‍🤝‍👨🏻","👩🏼‍🤝‍👨🏽","👩🏼‍🤝‍👨🏾","👩🏼‍🤝‍👨🏿","👩🏽‍🤝‍👨🏻","👩🏽‍🤝‍👨🏼","👩🏽‍🤝‍👨🏾","👩🏽‍🤝‍👨🏿","👩🏾‍🤝‍👨🏻","👩🏾‍🤝‍👨🏼","👩🏾‍🤝‍👨🏽","👩🏾‍🤝‍👨🏿","👩🏿‍🤝‍👨🏻","👩🏿‍🤝‍👨🏼","👩🏿‍🤝‍👨🏽","👩🏿‍🤝‍👨🏾","👨🏻‍🤝‍👨🏼","👨🏻‍🤝‍👨🏽","👨🏻‍🤝‍👨🏾","👨🏻‍🤝‍👨🏿","👨🏼‍🤝‍👨🏻","👨🏼‍🤝‍👨🏽","👨🏼‍🤝‍👨🏾","👨🏼‍🤝‍👨🏿","👨🏽‍🤝‍👨🏻","👨🏽‍🤝‍👨🏼","👨🏽‍🤝‍👨🏾","👨🏽‍🤝‍👨🏿","👨🏾‍🤝‍👨🏻","👨🏾‍🤝‍👨🏼","👨🏾‍🤝‍👨🏽","👨🏾‍🤝‍👨🏿","👨🏿‍🤝‍👨🏻","👨🏿‍🤝‍👨🏼","👨🏿‍🤝‍👨🏽","👨🏿‍🤝‍👨🏾","💏","🧑🏻‍❤️‍💋‍🧑🏼","🧑🏻‍❤️‍💋‍🧑🏽","🧑🏻‍❤️‍💋‍🧑🏾","🧑🏻‍❤️‍💋‍🧑🏿","🧑🏼‍❤️‍💋‍🧑🏻","🧑🏼‍❤️‍💋‍🧑🏽","🧑🏼‍❤️‍💋‍🧑🏾","🧑🏼‍❤️‍💋‍🧑🏿","🧑🏽‍❤️‍💋‍🧑🏻","🧑🏽‍❤️‍💋‍🧑🏼","🧑🏽‍❤️‍💋‍🧑🏾","🧑🏽‍❤️‍💋‍🧑🏿","🧑🏾‍❤️‍💋‍🧑🏻","🧑🏾‍❤️‍💋‍🧑🏼","🧑🏾‍❤️‍💋‍🧑🏽","🧑🏾‍❤️‍💋‍🧑🏿","🧑🏿‍❤️‍💋‍🧑🏻","🧑🏿‍❤️‍💋‍🧑🏼","🧑🏿‍❤️‍💋‍🧑🏽","🧑🏿‍❤️‍💋‍🧑🏾","👩‍❤️‍💋‍👨","👩🏻‍❤️‍💋‍👨🏻","👩🏻‍❤️‍💋‍👨🏼","👩🏻‍❤️‍💋‍👨🏽","👩🏻‍❤️‍💋‍👨🏾","👩🏻‍❤️‍💋‍👨🏿","👩🏼‍❤️‍💋‍👨🏻","👩🏼‍❤️‍💋‍👨🏼","👩🏼‍❤️‍💋‍👨🏽","👩🏼‍❤️‍💋‍👨🏾","👩🏼‍❤️‍💋‍👨🏿","👩🏽‍❤️‍💋‍👨🏻","👩🏽‍❤️‍💋‍👨🏼","👩🏽‍❤️‍💋‍👨🏽","👩🏽‍❤️‍💋‍👨🏾","👩🏽‍❤️‍💋‍👨🏿","👩🏾‍❤️‍💋‍👨🏻","👩🏾‍❤️‍💋‍👨🏼","👩🏾‍❤️‍💋‍👨🏽","👩🏾‍❤️‍💋‍👨🏾","👩🏾‍❤️‍💋‍👨🏿","👩🏿‍❤️‍💋‍👨🏻","👩🏿‍❤️‍💋‍👨🏼","👩🏿‍❤️‍💋‍👨🏽","👩🏿‍❤️‍💋‍👨🏾","👩🏿‍❤️‍💋‍👨🏿","👨‍❤️‍💋‍👨","👨🏻‍❤️‍💋‍👨🏻","👨🏻‍❤️‍💋‍👨🏼","👨🏻‍❤️‍💋‍👨🏽","👨🏻‍❤️‍💋‍👨🏾","👨🏻‍❤️‍💋‍👨🏿","👨🏼‍❤️‍💋‍👨🏻","👨🏼‍❤️‍💋‍👨🏼","👨🏼‍❤️‍💋‍👨🏽","👨🏼‍❤️‍💋‍👨🏾","👨🏼‍❤️‍💋‍👨🏿","👨🏽‍❤️‍💋‍👨🏻","👨🏽‍❤️‍💋‍👨🏼","👨🏽‍❤️‍💋‍👨🏽","👨🏽‍❤️‍💋‍👨🏾","👨🏽‍❤️‍💋‍👨🏿","👨🏾‍❤️‍💋‍👨🏻","👨🏾‍❤️‍💋‍👨🏼","👨🏾‍❤️‍💋‍👨🏽","👨🏾‍❤️‍💋‍👨🏾","👨🏾‍❤️‍💋‍👨🏿","👨🏿‍❤️‍💋‍👨🏻","👨🏿‍❤️‍💋‍👨🏼","👨🏿‍❤️‍💋‍👨🏽","👨🏿‍❤️‍💋‍👨🏾","👨🏿‍❤️‍💋‍👨🏿","👩‍❤️‍💋‍👩","👩🏻‍❤️‍💋‍👩🏻","👩🏻‍❤️‍💋‍👩🏼","👩🏻‍❤️‍💋‍👩🏽","👩🏻‍❤️‍💋‍👩🏾","👩🏻‍❤️‍💋‍👩🏿","👩🏼‍❤️‍💋‍👩🏻","👩🏼‍❤️‍💋‍👩🏼","👩🏼‍❤️‍💋‍👩🏽","👩🏼‍❤️‍💋‍👩🏾","👩🏼‍❤️‍💋‍👩🏿","👩🏽‍❤️‍💋‍👩🏻","👩🏽‍❤️‍💋‍👩🏼","👩🏽‍❤️‍💋‍👩🏽","👩🏽‍❤️‍💋‍👩🏾","👩🏽‍❤️‍💋‍👩🏿","👩🏾‍❤️‍💋‍👩🏻","👩🏾‍❤️‍💋‍👩🏼","👩🏾‍❤️‍💋‍👩🏽","👩🏾‍❤️‍💋‍👩🏾","👩🏾‍❤️‍💋‍👩🏿","👩🏿‍❤️‍💋‍👩🏻","👩🏿‍❤️‍💋‍👩🏼","👩🏿‍❤️‍💋‍👩🏽","👩🏿‍❤️‍💋‍👩🏾","👩🏿‍❤️‍💋‍👩🏿","💑","🧑🏻‍❤️‍🧑🏼","🧑🏻‍❤️‍🧑🏽","🧑🏻‍❤️‍🧑🏾","🧑🏻‍❤️‍🧑🏿","🧑🏼‍❤️‍🧑🏻","🧑🏼‍❤️‍🧑🏽","🧑🏼‍❤️‍🧑🏾","🧑🏼‍❤️‍🧑🏿","🧑🏽‍❤️‍🧑🏻","🧑🏽‍❤️‍🧑🏼","🧑🏽‍❤️‍🧑🏾","🧑🏽‍❤️‍🧑🏿","🧑🏾‍❤️‍🧑🏻","🧑🏾‍❤️‍🧑🏼","🧑🏾‍❤️‍🧑🏽","🧑🏾‍❤️‍🧑🏿","🧑🏿‍❤️‍🧑🏻","🧑🏿‍❤️‍🧑🏼","🧑🏿‍❤️‍🧑🏽","🧑🏿‍❤️‍🧑🏾","👩‍❤️‍👨","👩🏻‍❤️‍👨🏻","👩🏻‍❤️‍👨🏼","👩🏻‍❤️‍👨🏽","👩🏻‍❤️‍👨🏾","👩🏻‍❤️‍👨🏿","👩🏼‍❤️‍👨🏻","👩🏼‍❤️‍👨🏼","👩🏼‍❤️‍👨🏽","👩🏼‍❤️‍👨🏾","👩🏼‍❤️‍👨🏿","👩🏽‍❤️‍👨🏻","👩🏽‍❤️‍👨🏼","👩🏽‍❤️‍👨🏽","👩🏽‍❤️‍👨🏾","👩🏽‍❤️‍👨🏿","👩🏾‍❤️‍👨🏻","👩🏾‍❤️‍👨🏼","👩🏾‍❤️‍👨🏽","👩🏾‍❤️‍👨🏾","👩🏾‍❤️‍👨🏿","👩🏿‍❤️‍👨🏻","👩🏿‍❤️‍👨🏼","👩🏿‍❤️‍👨🏽","👩🏿‍❤️‍👨🏾","👩🏿‍❤️‍👨🏿","👨‍❤️‍👨","👨🏻‍❤️‍👨🏻","👨🏻‍❤️‍👨🏼","👨🏻‍❤️‍👨🏽","👨🏻‍❤️‍👨🏾","👨🏻‍❤️‍👨🏿","👨🏼‍❤️‍👨🏻","👨🏼‍❤️‍👨🏼","👨🏼‍❤️‍👨🏽","👨🏼‍❤️‍👨🏾","👨🏼‍❤️‍👨🏿","👨🏽‍❤️‍👨🏻","👨🏽‍❤️‍👨🏼","👨🏽‍❤️‍👨🏽","👨🏽‍❤️‍👨🏾","👨🏽‍❤️‍👨🏿","👨🏾‍❤️‍👨🏻","👨🏾‍❤️‍👨🏼","👨🏾‍❤️‍👨🏽","👨🏾‍❤️‍👨🏾","👨🏾‍❤️‍👨🏿","👨🏿‍❤️‍👨🏻","👨🏿‍❤️‍👨🏼","👨🏿‍❤️‍👨🏽","👨🏿‍❤️‍👨🏾","👨🏿‍❤️‍👨🏿","👩‍❤️‍👩","👩🏻‍❤️‍👩🏻","👩🏻‍❤️‍👩🏼","👩🏻‍❤️‍👩🏽","👩🏻‍❤️‍👩🏾","👩🏻‍❤️‍👩🏿","👩🏼‍❤️‍👩🏻","👩🏼‍❤️‍👩🏼","👩🏼‍❤️‍👩🏽","👩🏼‍❤️‍👩🏾","👩🏼‍❤️‍👩🏿","👩🏽‍❤️‍👩🏻","👩🏽‍❤️‍👩🏼","👩🏽‍❤️‍👩🏽","👩🏽‍❤️‍👩🏾","👩🏽‍❤️‍👩🏿","👩🏾‍❤️‍👩🏻","👩🏾‍❤️‍👩🏼","👩🏾‍❤️‍👩🏽","👩🏾‍❤️‍👩🏾","👩🏾‍❤️‍👩🏿","👩🏿‍❤️‍👩🏻","👩🏿‍❤️‍👩🏼","👩🏿‍❤️‍👩🏽","👩🏿‍❤️‍👩🏾","👩🏿‍❤️‍👩🏿","👪","👨‍👩‍👦","👨‍👩‍👧","👨‍👩‍👧‍👦","👨‍👩‍👦‍👦","👨‍👩‍👧‍👧","👨‍👨‍👦","👨‍👨‍👧","👨‍👨‍👧‍👦","👨‍👨‍👦‍👦","👨‍👨‍👧‍👧","👩‍👩‍👦","👩‍👩‍👧","👩‍👩‍👧‍👦","👩‍👩‍👦‍👦","👩‍👩‍👧‍👧","👨‍👦","👨‍👦‍👦","👨‍👧","👨‍👧‍👦","👨‍👧‍👧","👩‍👦","👩‍👦‍👦","👩‍👧","👩‍👧‍👦","👩‍👧‍👧","🐕‍🦺","🐈‍⬛","🐻‍❄️","🏳️‍🌈","🏳️‍⚧️","🏴‍☠️"],"expected":[{"id":"0","event":null,"data":"Blåbæ\nrsyltetøy må da være mulig å oppdrive? 😶‍🌫️ 😮‍💨 😵‍💫 ❤️‍🔥 ❤️‍🩹 👁️‍🗨️ 🫱🏻‍🫲🏼 🫱🏻‍🫲🏽 🫱🏻‍🫲🏾 🫱🏻‍🫲🏿 🫱🏼‍🫲🏻 🫱🏼‍🫲🏽 🫱🏼‍🫲🏾 🫱🏼‍🫲🏿 🫱🏽‍🫲🏻 🫱🏽‍🫲🏼 🫱🏽‍🫲🏾 🫱🏽‍🫲🏿 🫱🏾‍🫲🏻 🫱🏾‍🫲🏼 🫱🏾‍🫲🏽 🫱🏾‍🫲🏿 🫱🏿‍🫲🏻 🫱🏿‍🫲🏼 🫱🏿‍🫲🏽 🫱🏿‍🫲🏾 🧔‍♂️ 🧔🏻‍♂️ 🧔🏼‍♂️ 🧔🏽‍♂️ 🧔🏾‍♂️ 🧔🏿‍♂️ 🧔‍♀️ 🧔🏻‍♀️ 🧔🏼‍♀️ 🧔🏽‍♀️ 🧔🏾‍♀️ 🧔🏿‍♀️ 👨‍🦰 👨🏻‍🦰 👨🏼‍🦰 👨🏽‍🦰 👨🏾‍🦰 👨🏿‍🦰 👨‍🦱 👨🏻‍🦱 👨🏼‍🦱 👨🏽‍🦱 👨🏾‍🦱 👨🏿‍🦱 👨‍🦳 👨🏻‍🦳 👨🏼‍🦳 👨🏽‍🦳 👨🏾‍🦳 👨🏿‍🦳 👨‍🦲 👨🏻‍🦲 👨🏼‍🦲 👨🏽‍🦲 👨🏾‍🦲 👨🏿‍🦲 👩‍🦰 👩🏻‍🦰 👩🏼‍🦰 👩🏽‍🦰 👩🏾‍🦰 👩🏿‍🦰 🧑‍🦰 🧑🏻‍🦰 🧑🏼‍🦰 🧑🏽‍🦰 🧑🏾‍🦰 🧑🏿‍🦰 👩‍🦱 👩🏻‍🦱 👩🏼‍🦱 👩🏽‍🦱 👩🏾‍🦱 👩🏿‍🦱 🧑‍🦱 🧑🏻‍🦱 🧑🏼‍🦱 🧑🏽‍🦱 🧑🏾‍🦱 🧑🏿‍🦱 👩‍🦳 👩🏻‍🦳 👩🏼‍🦳 👩🏽‍🦳 👩🏾‍🦳 👩🏿‍🦳 🧑‍🦳 🧑🏻‍🦳 🧑🏼‍🦳 🧑🏽‍🦳 🧑🏾‍🦳"},{"id":"1","event":null,"data":"በእርግጥ አንድ ሰው ብሉቤሪ ጃምን መግዛት መቻል አለበት? 🧑🏿‍🦳 👩‍🦲 👩🏻‍🦲 👩🏼‍🦲 👩🏽‍🦲 👩🏾‍🦲 👩🏿‍🦲 🧑‍🦲 🧑🏻‍🦲 🧑🏼‍🦲 🧑🏽‍🦲 🧑🏾‍🦲 🧑🏿‍🦲 👱‍♀️ 👱🏻‍♀️ 👱🏼‍♀️ 👱🏽‍♀️ 👱🏾‍♀️ 👱🏿‍♀️ 👱‍♂️ 👱🏻‍♂️ 👱🏼‍♂️ 👱🏽‍♂️ 👱🏾‍♂️ 👱🏿‍♂️ 🙍‍♂️ 🙍🏻‍♂️ 🙍🏼‍♂️ 🙍🏽‍♂️ 🙍🏾‍♂️ 🙍🏿‍♂️ 🙍‍♀️ 🙍🏻‍♀️ 🙍🏼‍♀️ 🙍🏽‍♀️ 🙍🏾‍♀️ 🙍🏿‍♀️ 🙎‍♂️ 🙎🏻‍♂️ 🙎🏼‍♂️ 🙎🏽‍♂️ 🙎🏾‍♂️ 🙎🏿‍♂️ 🙎‍♀️ 🙎🏻‍♀️ 🙎🏼‍♀️ 🙎🏽‍♀️ 🙎🏾‍♀️ 🙎🏿‍♀️ 🙅‍♂️ 🙅🏻‍♂️ 🙅🏼‍♂️ 🙅🏽‍♂️ 🙅🏾‍♂️ 🙅🏿‍♂️ 🙅‍♀️ 🙅🏻‍♀️ 🙅🏼‍♀️ 🙅🏽‍♀️ 🙅🏾‍♀️ 🙅🏿‍♀️ 🙆‍♂️ 🙆🏻‍♂️ 🙆🏼‍♂️ 🙆🏽‍♂️ 🙆🏾‍♂️ 🙆🏿‍♂️ 🙆‍♀️ 🙆🏻‍♀️ 🙆🏼‍♀️ 🙆🏽‍♀️ 🙆🏾‍♀️ 🙆🏿‍♀️ 💁‍♂️ 💁🏻‍♂️ 💁🏼‍♂️ 💁🏽‍♂️ 💁🏾‍♂️ 💁🏿‍♂️ 💁‍♀️ 💁🏻‍♀️ 💁🏼‍♀️ 💁🏽‍♀️ 💁🏾‍♀️ 💁🏿‍♀️ 🙋‍♂️ 🙋🏻‍♂️ 🙋🏼‍♂️ 🙋🏽‍♂️ 🙋🏾‍♂️ 🙋🏿‍♂️ 🙋‍♀️ 🙋🏻‍♀️ 🙋🏼‍♀️ 🙋🏽‍♀️ 🙋🏾‍♀️ 🙋🏿‍♀️"},{"id":"2","event":null,"data":"بالتأ\nكيد يجب أن يكون المرء قادرا على شراء مربى التوت؟ 🧏‍♂️ 🧏🏻‍♂️ 🧏🏼‍♂️ 🧏🏽‍♂️ 🧏🏾‍♂️ 🧏🏿‍♂️ 🧏‍♀️ 🧏🏻‍♀️ 🧏🏼‍♀️ 🧏🏽‍♀️ 🧏🏾‍♀️ 🧏🏿‍♀️ 🙇‍♂️ 🙇🏻‍♂️ 🙇🏼‍♂️ 🙇🏽‍♂️ 🙇🏾‍♂️ 🙇🏿‍♂️ 🙇‍♀️ 🙇🏻‍♀️ 🙇🏼‍♀️ 🙇🏽‍♀️ 🙇🏾‍♀️ 🙇🏿‍♀️ 🤦‍♂️ 🤦🏻‍♂️ 🤦🏼‍♂️ 🤦🏽‍♂️ 🤦🏾‍♂️ 🤦🏿‍♂️ 🤦‍♀️ 🤦🏻‍♀️ 🤦🏼‍♀️ 🤦🏽‍♀️ 🤦🏾‍♀️ 🤦🏿‍♀️ 🤷‍♂️ 🤷🏻‍♂️ 🤷🏼‍♂️ 🤷🏽‍♂️ 🤷🏾‍♂️ 🤷🏿‍♂️ 🤷‍♀️ 🤷🏻‍♀️ 🤷🏼‍♀️ 🤷🏽‍♀️ 🤷🏾‍♀️ 🤷🏿‍♀️ 🧑‍⚕️ 🧑🏻‍⚕️ 🧑🏼‍⚕️ 🧑🏽‍⚕️ 🧑🏾‍⚕️ 🧑🏿‍⚕️ 👨‍⚕️ 👨🏻‍⚕️ 👨🏼‍⚕️ 👨🏽‍⚕️ 👨🏾‍⚕️ 👨🏿‍⚕️ 👩‍⚕️ 👩🏻‍⚕️ 👩🏼‍⚕️ 👩🏽‍⚕️ 👩🏾‍⚕️ 👩🏿‍⚕️ 🧑‍🎓 🧑🏻‍🎓 🧑🏼‍🎓 🧑🏽‍🎓 🧑🏾‍🎓 🧑🏿‍🎓 👨‍🎓 👨🏻‍🎓 👨🏼‍🎓 👨🏽‍🎓 👨🏾‍🎓 👨🏿‍🎓 👩‍🎓 👩🏻‍🎓 👩🏼‍🎓 👩🏽‍🎓 👩🏾‍🎓 👩🏿‍🎓 🧑‍🏫 🧑🏻‍🏫 🧑🏼‍🏫 🧑🏽‍🏫 🧑🏾‍🏫 🧑🏿‍🏫 👨‍🏫 👨🏻‍🏫 👨🏼‍🏫 👨🏽‍🏫 👨🏾‍🏫 👨🏿‍🏫 👩‍🏫"},{"id":"3","event":null,"data":"一定能买到蓝莓果酱吗? 👩🏻‍🏫 👩🏼‍🏫 👩🏽‍🏫 👩🏾‍🏫 👩🏿‍🏫 🧑‍⚖️ 🧑🏻‍⚖️ 🧑🏼‍⚖️ 🧑🏽‍⚖️ 🧑🏾‍⚖️ 🧑🏿‍⚖️ 👨‍⚖️ 👨🏻‍⚖️ 👨🏼‍⚖️ 👨🏽‍⚖️ 👨🏾‍⚖️ 👨🏿‍⚖️ 👩‍⚖️ 👩🏻‍⚖️ 👩🏼‍⚖️ 👩🏽‍⚖️ 👩🏾‍⚖️ 👩🏿‍⚖️ 🧑‍🌾 🧑🏻‍🌾 🧑🏼‍🌾 🧑🏽‍🌾 🧑🏾‍🌾 🧑🏿‍🌾 👨‍🌾 👨🏻‍🌾 👨🏼‍🌾 👨🏽‍🌾 👨🏾‍🌾 👨🏿‍🌾 👩‍🌾 👩🏻‍🌾 👩🏼‍🌾 👩🏽‍🌾 👩🏾‍🌾 👩🏿‍🌾 🧑‍🍳 🧑🏻‍🍳 🧑🏼‍🍳 🧑🏽‍🍳 🧑🏾‍🍳 🧑🏿‍🍳 👨‍🍳 👨🏻‍🍳 👨🏼‍🍳 👨🏽‍🍳 👨🏾‍🍳 👨🏿‍🍳 👩‍🍳 👩🏻‍🍳 👩🏼‍🍳 👩🏽‍🍳 👩🏾‍🍳 👩🏿‍🍳 🧑‍🔧 🧑🏻‍🔧 🧑🏼‍🔧 🧑🏽‍🔧 🧑🏾‍🔧 🧑🏿‍🔧 👨‍🔧 👨🏻‍🔧 👨🏼‍🔧 👨🏽‍🔧 👨🏾‍🔧 👨🏿‍🔧 👩‍🔧 👩🏻‍🔧 👩🏼‍🔧 👩🏽‍🔧 👩🏾‍🔧 👩🏿‍🔧 🧑‍🏭 🧑🏻‍🏭 🧑🏼‍🏭 🧑🏽‍🏭 🧑🏾‍🏭 🧑🏿‍🏭 👨‍🏭 👨🏻‍🏭 👨🏼‍🏭 👨🏽‍🏭 👨🏾‍🏭 👨🏿‍🏭 👩‍🏭 👩🏻‍🏭 👩🏼‍🏭 👩🏽‍🏭 👩🏾‍🏭 👩🏿‍🏭 🧑‍💼 🧑🏻‍💼"},{"id":"4","event":null,"data":"ચોક્ક\nસ એક બ્લુબેરી જામ મેળવવા માટે સમર્થ હોવા જ જોઈએ? 🧑🏼‍💼 🧑🏽‍💼 🧑🏾‍💼 🧑🏿‍💼 👨‍💼 👨🏻‍💼 👨🏼‍💼 👨🏽‍💼 👨🏾‍💼 👨🏿‍💼 👩‍💼 👩🏻‍💼 👩🏼‍💼 👩🏽‍💼 👩🏾‍💼 👩🏿‍💼 🧑‍🔬 🧑🏻‍🔬 🧑🏼‍🔬 🧑🏽‍🔬 🧑🏾‍🔬 🧑🏿‍🔬 👨‍🔬 👨🏻‍🔬 👨🏼‍🔬 👨🏽‍🔬 👨🏾‍🔬 👨🏿‍🔬 👩‍🔬 👩🏻‍🔬 👩🏼‍🔬 👩🏽‍🔬 👩🏾‍🔬 👩🏿‍🔬 🧑‍💻 🧑🏻‍💻 🧑🏼‍💻 🧑🏽‍💻 🧑🏾‍💻 🧑🏿‍💻 👨‍💻 👨🏻‍💻 👨🏼‍💻 👨🏽‍💻 👨🏾‍💻 👨🏿‍💻 👩‍💻 👩🏻‍💻 👩🏼‍💻 👩🏽‍💻 👩🏾‍💻 👩🏿‍💻 🧑‍🎤 🧑🏻‍🎤 🧑🏼‍🎤 🧑🏽‍🎤 🧑🏾‍🎤 🧑🏿‍🎤 👨‍🎤 👨🏻‍🎤 👨🏼‍🎤 👨🏽‍🎤 👨🏾‍🎤 👨🏿‍🎤 👩‍🎤 👩🏻‍🎤 👩🏼‍🎤 👩🏽‍🎤 👩🏾‍🎤 👩🏿‍🎤 🧑‍🎨 🧑🏻‍🎨 🧑🏼‍🎨 🧑🏽‍🎨 🧑🏾‍🎨 🧑🏿‍🎨 👨‍🎨 👨🏻‍🎨 👨🏼‍🎨 👨🏽‍🎨 👨🏾‍🎨 👨🏿‍🎨 👩‍🎨 👩🏻‍🎨 👩🏼‍🎨 👩🏽‍🎨 👩🏾‍🎨 👩🏿‍🎨 🧑‍✈️ 🧑🏻‍✈️ 🧑🏼‍✈️ 🧑🏽‍✈️ 🧑🏾‍✈️ 🧑🏿‍✈️ 👨‍✈️ 👨🏻‍✈️ 👨🏼‍✈️"},{"id":"5","event":null,"data":"בטח אפשר להשיג ריבת אוכמניות? 👨🏽‍✈️ 👨🏾‍✈️ 👨🏿‍✈️ 👩‍✈️ 👩🏻‍✈️ 👩🏼‍✈️ 👩🏽‍✈️ 👩🏾‍✈️ 👩🏿‍✈️ 🧑‍🚀 🧑🏻‍🚀 🧑🏼‍🚀 🧑🏽‍🚀 🧑🏾‍🚀 🧑🏿‍🚀 👨‍🚀 👨🏻‍🚀 👨🏼‍🚀 👨🏽‍🚀 👨🏾‍🚀 👨🏿‍🚀 👩‍🚀 👩🏻‍🚀 👩🏼‍🚀 👩🏽‍🚀 👩🏾‍🚀 👩🏿‍🚀 🧑‍🚒 🧑🏻‍🚒 🧑🏼‍🚒 🧑🏽‍🚒 🧑🏾‍🚒 🧑🏿‍🚒 👨‍🚒 👨🏻‍🚒 👨🏼‍🚒 👨🏽‍🚒 👨🏾‍🚒 👨🏿‍🚒 👩‍🚒 👩🏻‍🚒 👩🏼‍🚒 👩🏽‍🚒 👩🏾‍🚒 👩🏿‍🚒 👮‍♂️ 👮🏻‍♂️ 👮🏼‍♂️ 👮🏽‍♂️ 👮🏾‍♂️ 👮🏿‍♂️ 👮‍♀️ 👮🏻‍♀️ 👮🏼‍♀️ 👮🏽‍♀️ 👮🏾‍♀️ 👮🏿‍♀️ 🕵️‍♂️ 🕵🏻‍♂️ 🕵🏼‍♂️ 🕵🏽‍♂️ 🕵🏾‍♂️ 🕵🏿‍♂️ 🕵️‍♀️ 🕵🏻‍♀️ 🕵🏼‍♀️ 🕵🏽‍♀️ 🕵🏾‍♀️ 🕵🏿‍♀️ 💂‍♂️ 💂🏻‍♂️ 💂🏼‍♂️ 💂🏽‍♂️ 💂🏾‍♂️ 💂🏿‍♂️ 💂‍♀️ 💂🏻‍♀️ 💂🏼‍♀️ 💂🏽‍♀️ 💂🏾‍♀️ 💂🏿‍♀️ 👷‍♂️ 👷🏻‍♂️ 👷🏼‍♂️ 👷🏽‍♂️ 👷🏾‍♂️ 👷🏿‍♂️ 👷‍♀️ 👷🏻‍♀️ 👷🏼‍♀️ 👷🏽‍♀️ 👷🏾‍♀️ 👷🏿‍♀️ 👳‍♂️ 👳🏻‍♂️ 👳🏼‍♂️ 👳🏽‍♂️"},{"id":"6","event":null,"data":"निश्च\nित रूप से किसी को ब्लूबेरी जैम खरीदने में सक्षम होना चाहिए? 👳🏾‍♂️ 👳🏿‍♂️ 👳‍♀️ 👳🏻‍♀️ 👳🏼‍♀️ 👳🏽‍♀️ 👳🏾‍♀️ 👳🏿‍♀️ 🤵‍♂️ 🤵🏻‍♂️ 🤵🏼‍♂️ 🤵🏽‍♂️ 🤵🏾‍♂️ 🤵🏿‍♂️ 🤵‍♀️ 🤵🏻‍♀️ 🤵🏼‍♀️ 🤵🏽‍♀️ 🤵🏾‍♀️ 🤵🏿‍♀️ 👰‍♂️ 👰🏻‍♂️ 👰🏼‍♂️ 👰🏽‍♂️ 👰🏾‍♂️ 👰🏿‍♂️ 👰‍♀️ 👰🏻‍♀️ 👰🏼‍♀️ 👰🏽‍♀️ 👰🏾‍♀️ 👰🏿‍♀️ 👩‍🍼 👩🏻‍🍼 👩🏼‍🍼 👩🏽‍🍼 👩🏾‍🍼 👩🏿‍🍼 👨‍🍼 👨🏻‍🍼 👨🏼‍🍼 👨🏽‍🍼 👨🏾‍🍼 👨🏿‍🍼 🧑‍🍼 🧑🏻‍🍼 🧑🏼‍🍼 🧑🏽‍🍼 🧑🏾‍🍼 🧑🏿‍🍼 🧑‍🎄 🧑🏻‍🎄 🧑🏼‍🎄 🧑🏽‍🎄 🧑🏾‍🎄 🧑🏿‍🎄 🦸‍♂️ 🦸🏻‍♂️ 🦸🏼‍♂️ 🦸🏽‍♂️ 🦸🏾‍♂️ 🦸🏿‍♂️ 🦸‍♀️ 🦸🏻‍♀️ 🦸🏼‍♀️ 🦸🏽‍♀️ 🦸🏾‍♀️ 🦸🏿‍♀️ 🦹‍♂️ 🦹🏻‍♂️ 🦹🏼‍♂️ 🦹🏽‍♂️ 🦹🏾‍♂️ 🦹🏿‍♂️ 🦹‍♀️ 🦹🏻‍♀️ 🦹🏼‍♀️ 🦹🏽‍♀️ 🦹🏾‍♀️ 🦹🏿‍♀️ 🧙‍♂️ 🧙🏻‍♂️ 🧙🏼‍♂️ 🧙🏽‍♂️ 🧙🏾‍♂️ 🧙🏿‍♂️ 🧙‍♀️ 🧙🏻‍♀️ 🧙🏼‍♀️ 🧙🏽‍♀️ 🧙🏾‍♀️ 🧙🏿‍♀️ 🧚‍♂️ 🧚🏻‍♂️ 🧚🏼‍♂️ 🧚🏽‍♂️ 🧚🏾‍♂️"},{"id":"7","event":null,"data":"Bê guman pêdivî ye ku meriv karibe jama şînê peyda bike? 🧚🏿‍♂️ 🧚‍♀️ 🧚🏻‍♀️ 🧚🏼‍♀️ 🧚🏽‍♀️ 🧚🏾‍♀️ 🧚🏿‍♀️ 🧛‍♂️ 🧛🏻‍♂️ 🧛🏼‍♂️ 🧛🏽‍♂️ 🧛🏾‍♂️ 🧛🏿‍♂️ 🧛‍♀️ 🧛🏻‍♀️ 🧛🏼‍♀️ 🧛🏽‍♀️ 🧛🏾‍♀️ 🧛🏿‍♀️ 🧜‍♂️ 🧜🏻‍♂️ 🧜🏼‍♂️ 🧜🏽‍♂️ 🧜🏾‍♂️ 🧜🏿‍♂️ 🧜‍♀️ 🧜🏻‍♀️ 🧜🏼‍♀️ 🧜🏽‍♀️ 🧜🏾‍♀️ 🧜🏿‍♀️ 🧝‍♂️ 🧝🏻‍♂️ 🧝🏼‍♂️ 🧝🏽‍♂️ 🧝🏾‍♂️ 🧝🏿‍♂️ 🧝‍♀️ 🧝🏻‍♀️ 🧝🏼‍♀️ 🧝🏽‍♀️ 🧝🏾‍♀️ 🧝🏿‍♀️ 🧞‍♂️ 🧞‍♀️ 🧟‍♂️ 🧟‍♀️ 💆‍♂️ 💆🏻‍♂️ 💆🏼‍♂️ 💆🏽‍♂️ 💆🏾‍♂️ 💆🏿‍♂️ 💆‍♀️ 💆🏻‍♀️ 💆🏼‍♀️ 💆🏽‍♀️ 💆🏾‍♀️ 💆🏿‍♀️ 💇‍♂️ 💇🏻‍♂️ 💇🏼‍♂️ 💇🏽‍♂️ 💇🏾‍♂️ 💇🏿‍♂️ 💇‍♀️ 💇🏻‍♀️ 💇🏼‍♀️ 💇🏽‍♀️ 💇🏾‍♀️ 💇🏿‍♀️ 🚶‍♂️ 🚶🏻‍♂️ 🚶🏼‍♂️ 🚶🏽‍♂️ 🚶🏾‍♂️ 🚶🏿‍♂️ 🚶‍♀️ 🚶🏻‍♀️ 🚶🏼‍♀️ 🚶🏽‍♀️ 🚶🏾‍♀️ 🚶🏿‍♀️ 🧍‍♂️ 🧍🏻‍♂️ 🧍🏼‍♂️ 🧍🏽‍♂️ 🧍🏾‍♂️ 🧍🏿‍♂️ 🧍‍♀️ 🧍🏻‍♀️ 🧍🏼‍♀️ 🧍🏽‍♀️ 🧍🏾‍♀️ 🧍🏿‍♀️ 🧎‍♂️ 🧎🏻‍♂️"},{"id":"8","event":null,"data":"ប្រាក\nដណាស់ មនុស្សម្នាក់ត្រូវតែអាចទិញយៈសាពូនមី blueberry បានទេ? 🧎🏼‍♂️ 🧎🏽‍♂️ 🧎🏾‍♂️ 🧎🏿‍♂️ 🧎‍♀️ 🧎🏻‍♀️ 🧎🏼‍♀️ 🧎🏽‍♀️ 🧎🏾‍♀️ 🧎🏿‍♀️ 🧑‍🦯 🧑🏻‍🦯 🧑🏼‍🦯 🧑🏽‍🦯 🧑🏾‍🦯 🧑🏿‍🦯 👨‍🦯 👨🏻‍🦯 👨🏼‍🦯 👨🏽‍🦯 👨🏾‍🦯 👨🏿‍🦯 👩‍🦯 👩🏻‍🦯 👩🏼‍🦯 👩🏽‍🦯 👩🏾‍🦯 👩🏿‍🦯 🧑‍🦼 🧑🏻‍🦼 🧑🏼‍🦼 🧑🏽‍🦼 🧑🏾‍🦼 🧑🏿‍🦼 👨‍🦼 👨🏻‍🦼 👨🏼‍🦼 👨🏽‍🦼 👨🏾‍🦼 👨🏿‍🦼 👩‍🦼 👩🏻‍🦼 👩🏼‍🦼 👩🏽‍🦼 👩🏾‍🦼 👩🏿‍🦼 🧑‍🦽 🧑🏻‍🦽 🧑🏼‍🦽 🧑🏽‍🦽 🧑🏾‍🦽 🧑🏿‍🦽 👨‍🦽 👨🏻‍🦽 👨🏼‍🦽 👨🏽‍🦽 👨🏾‍🦽 👨🏿‍🦽 👩‍🦽 👩🏻‍🦽 👩🏼‍🦽 👩🏽‍🦽 👩🏾‍🦽 👩🏿‍🦽 🏃‍♂️ 🏃🏻‍♂️ 🏃🏼‍♂️ 🏃🏽‍♂️ 🏃🏾‍♂️ 🏃🏿‍♂️ 🏃‍♀️ 🏃🏻‍♀️ 🏃🏼‍♀️ 🏃🏽‍♀️ 🏃🏾‍♀️ 🏃🏿‍♀️ 👯‍♂️ 👯‍♀️ 🧖‍♂️ 🧖🏻‍♂️ 🧖🏼‍♂️ 🧖🏽‍♂️ 🧖🏾‍♂️ 🧖🏿‍♂️ 🧖‍♀️ 🧖🏻‍♀️ 🧖🏼‍♀️ 🧖🏽‍♀️ 🧖🏾‍♀️ 🧖🏿‍♀️ 🧗‍♂️ 🧗🏻‍♂️ 🧗🏼‍♂️ 🧗🏽‍♂️ 🧗🏾‍♂️ 🧗🏿‍♂️ 🧗‍♀️"},{"id":"9","event":null,"data":"ಖಂಡಿತವಾಗಿಯೂ ಒಬ್ಬರು ಬ್ಲೂಬೆರ್ರಿ ಜಾಮ್ ಅನ್ನು ಸಂಗ್ರಹಿಸಲು ಶಕ್ತರಾಗಿರಬೇಕು? 🧗🏻‍♀️ 🧗🏼‍♀️ 🧗🏽‍♀️ 🧗🏾‍♀️ 🧗🏿‍♀️ 🏌️‍♂️ 🏌🏻‍♂️ 🏌🏼‍♂️ 🏌🏽‍♂️ 🏌🏾‍♂️ 🏌🏿‍♂️ 🏌️‍♀️ 🏌🏻‍♀️ 🏌🏼‍♀️ 🏌🏽‍♀️ 🏌🏾‍♀️ 🏌🏿‍♀️ 🏄‍♂️ 🏄🏻‍♂️ 🏄🏼‍♂️ 🏄🏽‍♂️ 🏄🏾‍♂️ 🏄🏿‍♂️ 🏄‍♀️ 🏄🏻‍♀️ 🏄🏼‍♀️ 🏄🏽‍♀️ 🏄🏾‍♀️ 🏄🏿‍♀️ 🚣‍♂️ 🚣🏻‍♂️ 🚣🏼‍♂️ 🚣🏽‍♂️ 🚣🏾‍♂️ 🚣🏿‍♂️ 🚣‍♀️ 🚣🏻‍♀️ 🚣🏼‍♀️ 🚣🏽‍♀️ 🚣🏾‍♀️ 🚣🏿‍♀️ 🏊‍♂️ 🏊🏻‍♂️ 🏊🏼‍♂️ 🏊🏽‍♂️ 🏊🏾‍♂️ 🏊🏿‍♂️ 🏊‍♀️ 🏊🏻‍♀️ 🏊🏼‍♀️ 🏊🏽‍♀️ 🏊🏾‍♀️ 🏊🏿‍♀️ ⛹️‍♂️ ⛹🏻‍♂️ ⛹🏼‍♂️ ⛹🏽‍♂️ ⛹🏾‍♂️ ⛹🏿‍♂️ ⛹️‍♀️ ⛹🏻‍♀️ ⛹🏼‍♀️ ⛹🏽‍♀️ ⛹🏾‍♀️ ⛹🏿‍♀️ 🏋️‍♂️ 🏋🏻‍♂️ 🏋🏼‍♂️ 🏋🏽‍♂️ 🏋🏾‍♂️ 🏋🏿‍♂️ 🏋️‍♀️ 🏋🏻‍♀️ 🏋🏼‍♀️ 🏋🏽‍♀️ 🏋🏾‍♀️ 🏋🏿‍♀️ 🚴‍♂️ 🚴🏻‍♂️ 🚴🏼‍♂️ 🚴🏽‍♂️ 🚴🏾‍♂️ 🚴🏿‍♂️ 🚴‍♀️ 🚴🏻‍♀️ 🚴🏼‍♀️ 🚴🏽‍♀️ 🚴🏾‍♀️ 🚴🏿‍♀️ 🚵‍♂️ 🚵🏻‍♂️ 🚵🏼‍♂️ 🚵🏽‍♂️ 🚵🏾‍♂️ 🚵🏿‍♂️ 🚵‍♀️ 🚵🏻‍♀️"},{"id":"10","event":null,"data":"確かにブル\nーベリージャムを調達できなければなりませんか? 🚵🏼‍♀️ 🚵🏽‍♀️ 🚵🏾‍♀️ 🚵🏿‍♀️ 🤸‍♂️ 🤸🏻‍♂️ 🤸🏼‍♂️ 🤸🏽‍♂️ 🤸🏾‍♂️ 🤸🏿‍♂️ 🤸‍♀️ 🤸🏻‍♀️ 🤸🏼‍♀️ 🤸🏽‍♀️ 🤸🏾‍♀️ 🤸🏿‍♀️ 🤼‍♂️ 🤼‍♀️ 🤽‍♂️ 🤽🏻‍♂️ 🤽🏼‍♂️ 🤽🏽‍♂️ 🤽🏾‍♂️ 🤽🏿‍♂️ 🤽‍♀️ 🤽🏻‍♀️ 🤽🏼‍♀️ 🤽🏽‍♀️ 🤽🏾‍♀️ 🤽🏿‍♀️ 🤾‍♂️ 🤾🏻‍♂️ 🤾🏼‍♂️ 🤾🏽‍♂️ 🤾🏾‍♂️ 🤾🏿‍♂️ 🤾‍♀️ 🤾🏻‍♀️ 🤾🏼‍♀️ 🤾🏽‍♀️ 🤾🏾‍♀️ 🤾🏿‍♀️ 🤹‍♂️ 🤹🏻‍♂️ 🤹🏼‍♂️ 🤹🏽‍♂️ 🤹🏾‍♂️ 🤹🏿‍♂️ 🤹‍♀️ 🤹🏻‍♀️ 🤹🏼‍♀️ 🤹🏽‍♀️ 🤹🏾‍♀️ 🤹🏿‍♀️ 🧘‍♂️ 🧘🏻‍♂️ 🧘🏼‍♂️ 🧘🏽‍♂️ 🧘🏾‍♂️ 🧘🏿‍♂️ 🧘‍♀️ 🧘🏻‍♀️ 🧘🏼‍♀️ 🧘🏽‍♀️ 🧘🏾‍♀️ 🧘🏿‍♀️ 🧑‍🤝‍🧑 🧑🏻‍🤝‍🧑🏻 🧑🏻‍🤝‍🧑🏼 🧑🏻‍🤝‍🧑🏽 🧑🏻‍🤝‍🧑🏾 🧑🏻‍🤝‍🧑🏿 🧑🏼‍🤝‍🧑🏻 🧑🏼‍🤝‍🧑🏼 🧑🏼‍🤝‍🧑🏽 🧑🏼‍🤝‍🧑🏾 🧑🏼‍🤝‍🧑🏿 🧑🏽‍🤝‍🧑🏻 🧑🏽‍🤝‍🧑🏼 🧑🏽‍🤝‍🧑🏽 🧑🏽‍🤝‍🧑🏾 🧑🏽‍🤝‍🧑🏿 🧑🏾‍🤝‍🧑🏻 🧑🏾‍🤝‍🧑🏼 🧑🏾‍🤝‍🧑🏽 🧑🏾‍🤝‍🧑🏾 🧑🏾‍🤝‍🧑🏿 🧑🏿‍🤝‍🧑🏻 🧑🏿‍🤝‍🧑🏼 🧑🏿‍🤝‍🧑🏽 🧑🏿‍🤝‍🧑🏾 🧑🏿‍🤝‍🧑🏿 👩🏻‍🤝‍👩🏼 👩🏻‍🤝‍👩🏽 👩🏻‍🤝‍👩🏾 👩🏻‍🤝‍👩🏿 👩🏼‍🤝‍👩🏻"},{"id":"11","event":null,"data":"Невже треба вміти добути варення з чорниці? 👩🏼‍🤝‍👩🏽 👩🏼‍🤝‍👩🏾 👩🏼‍🤝‍👩🏿 👩🏽‍🤝‍👩🏻 👩🏽‍🤝‍👩🏼 👩🏽‍🤝‍👩🏾 👩🏽‍🤝‍👩🏿 👩🏾‍🤝‍👩🏻 👩🏾‍🤝‍👩🏼 👩🏾‍🤝‍👩🏽 👩🏾‍🤝‍👩🏿 👩🏿‍🤝‍👩🏻 👩🏿‍🤝‍👩🏼 👩🏿‍🤝‍👩🏽 👩🏿‍🤝‍👩🏾 👩🏻‍🤝‍👨🏼 👩🏻‍🤝‍👨🏽 👩🏻‍🤝‍👨🏾 👩🏻‍🤝‍👨🏿 👩🏼‍🤝‍👨🏻 👩🏼‍🤝‍👨🏽 👩🏼‍🤝‍👨🏾 👩🏼‍🤝‍👨🏿 👩🏽‍🤝‍👨🏻 👩🏽‍🤝‍👨🏼 👩🏽‍🤝‍👨🏾 👩🏽‍🤝‍👨🏿 👩🏾‍🤝‍👨🏻 👩🏾‍🤝‍👨🏼 👩🏾‍🤝‍👨🏽 👩🏾‍🤝‍👨🏿 👩🏿‍🤝‍👨🏻 👩🏿‍🤝‍👨🏼 👩🏿‍🤝‍👨🏽 👩🏿‍🤝‍👨🏾 👨🏻‍🤝‍👨🏼 👨🏻‍🤝‍👨🏽 👨🏻‍🤝‍👨🏾 👨🏻‍🤝‍👨🏿 👨🏼‍🤝‍👨🏻 👨🏼‍🤝‍👨🏽 👨🏼‍🤝‍👨🏾 👨🏼‍🤝‍👨🏿 👨🏽‍🤝‍👨🏻 👨🏽‍🤝‍👨🏼 👨🏽‍🤝‍👨🏾 👨🏽‍🤝‍👨🏿 👨🏾‍🤝‍👨🏻 👨🏾‍🤝‍👨🏼 👨🏾‍🤝‍👨🏽 👨🏾‍🤝‍👨🏿 👨🏿‍🤝‍👨🏻 👨🏿‍🤝‍👨🏼 👨🏿‍🤝‍👨🏽 👨🏿‍🤝‍👨🏾 💏 🧑🏻‍❤️‍💋‍🧑🏼 🧑🏻‍❤️‍💋‍🧑🏽 🧑🏻‍❤️‍💋‍🧑🏾 🧑🏻‍❤️‍💋‍🧑🏿 🧑🏼‍❤️‍💋‍🧑🏻 🧑🏼‍❤️‍💋‍🧑🏽 🧑🏼‍❤️‍💋‍🧑🏾 🧑🏼‍❤️‍💋‍🧑🏿 🧑🏽‍❤️‍💋‍🧑🏻 🧑🏽‍❤️‍💋‍🧑🏼 🧑🏽‍❤️‍💋‍🧑🏾 🧑🏽‍❤️‍💋‍🧑🏿 🧑🏾‍❤️‍💋‍🧑🏻 🧑🏾‍❤️‍💋‍🧑🏼 🧑🏾‍❤️‍💋‍🧑🏽 🧑🏾‍❤️‍💋‍🧑🏿 🧑🏿‍❤️‍💋‍🧑🏻 🧑🏿‍❤️‍💋‍🧑🏼 🧑🏿‍❤️‍💋‍🧑🏽 🧑🏿‍❤️‍💋‍🧑🏾 👩‍❤️‍💋‍👨 👩🏻‍❤️‍💋‍👨🏻 👩🏻‍❤️‍💋‍👨🏼 👩🏻‍❤️‍💋‍👨🏽 👩🏻‍❤️‍💋‍👨🏾 👩🏻‍❤️‍💋‍👨🏿 👩🏼‍❤️‍💋‍👨🏻 👩🏼‍❤️‍💋‍👨🏼 👩🏼‍❤️‍💋‍👨🏽 👩🏼‍❤️‍💋‍👨🏾 👩🏼‍❤️‍💋‍👨🏿 👩🏽‍❤️‍💋‍👨🏻 👩🏽‍❤️‍💋‍👨🏼 👩🏽‍❤️‍💋‍👨🏽 👩🏽‍❤️‍💋‍👨🏾 👩🏽‍❤️‍💋‍👨🏿 👩🏾‍❤️‍💋‍👨🏻 👩🏾‍❤️‍💋‍👨🏼 👩🏾‍❤️‍💋‍👨🏽 👩🏾‍❤️‍💋‍👨🏾 👩🏾‍❤️‍💋‍👨🏿"},{"id":"12","event":null,"data":"แน่นอ\nนว่าจะต้องสามารถจัดหาแยมบลูเบอร์รี่ได้? 👩🏿‍❤️‍💋‍👨🏻 👩🏿‍❤️‍💋‍👨🏼 👩🏿‍❤️‍💋‍👨🏽 👩🏿‍❤️‍💋‍👨🏾 👩🏿‍❤️‍💋‍👨🏿 👨‍❤️‍💋‍👨 👨🏻‍❤️‍💋‍👨🏻 👨🏻‍❤️‍💋‍👨🏼 👨🏻‍❤️‍💋‍👨🏽 👨🏻‍❤️‍💋‍👨🏾 👨🏻‍❤️‍💋‍👨🏿 👨🏼‍❤️‍💋‍👨🏻 👨🏼‍❤️‍💋‍👨🏼 👨🏼‍❤️‍💋‍👨🏽 👨🏼‍❤️‍💋‍👨🏾 👨🏼‍❤️‍💋‍👨🏿 👨🏽‍❤️‍💋‍👨🏻 👨🏽‍❤️‍💋‍👨🏼 👨🏽‍❤️‍💋‍👨🏽 👨🏽‍❤️‍💋‍👨🏾 👨🏽‍❤️‍💋‍👨🏿 👨🏾‍❤️‍💋‍👨🏻 👨🏾‍❤️‍💋‍👨🏼 👨🏾‍❤️‍💋‍👨🏽 👨🏾‍❤️‍💋‍👨🏾 👨🏾‍❤️‍💋‍👨🏿 👨🏿‍❤️‍💋‍👨🏻 👨🏿‍❤️‍💋‍👨🏼 👨🏿‍❤️‍💋‍👨🏽 👨🏿‍❤️‍💋‍👨🏾 👨🏿‍❤️‍💋‍👨🏿 👩‍❤️‍💋‍👩 👩🏻‍❤️‍💋‍👩🏻 👩🏻‍❤️‍💋‍👩🏼 👩🏻‍❤️‍💋‍👩🏽 👩🏻‍❤️‍💋‍👩🏾 👩🏻‍❤️‍💋‍👩🏿 👩🏼‍❤️‍💋‍👩🏻 👩🏼‍❤️‍💋‍👩🏼 👩🏼‍❤️‍💋‍👩🏽 👩🏼‍❤️‍💋‍👩🏾 👩🏼‍❤️‍💋‍👩🏿 👩🏽‍❤️‍💋‍👩🏻 👩🏽‍❤️‍💋‍👩🏼 👩🏽‍❤️‍💋‍👩🏽 👩🏽‍❤️‍💋‍👩🏾 👩🏽‍❤️‍💋‍👩🏿 👩🏾‍❤️‍💋‍👩🏻 👩🏾‍❤️‍💋‍👩🏼 👩🏾‍❤️‍💋‍👩🏽 👩🏾‍❤️‍💋‍👩🏾 👩🏾‍❤️‍💋‍👩🏿 👩🏿‍❤️‍💋‍👩🏻 👩🏿‍❤️‍💋‍👩🏼 👩🏿‍❤️‍💋‍👩🏽 👩🏿‍❤️‍💋‍👩🏾 👩🏿‍❤️‍💋‍👩🏿 💑 🧑🏻‍❤️‍🧑🏼 🧑🏻‍❤️‍🧑🏽 🧑🏻‍❤️‍🧑🏾 🧑🏻‍❤️‍🧑🏿 🧑🏼‍❤️‍🧑🏻 🧑🏼‍❤️‍🧑🏽 🧑🏼‍❤️‍🧑🏾 🧑🏼‍❤️‍🧑🏿 🧑🏽‍❤️‍🧑🏻 🧑🏽‍❤️‍🧑🏼 🧑🏽‍❤️‍🧑🏾 🧑🏽‍❤️‍🧑🏿 🧑🏾‍❤️‍🧑🏻 🧑🏾‍❤️‍🧑🏼 🧑🏾‍❤️‍🧑🏽 🧑🏾‍❤️‍🧑🏿 🧑🏿‍❤️‍🧑🏻 🧑🏿‍❤️‍🧑🏼 🧑🏿‍❤️‍🧑🏽 🧑🏿‍❤️‍🧑🏾 👩‍❤️‍👨 👩🏻‍❤️‍👨🏻 👩🏻‍❤️‍👨🏼 👩🏻‍❤️‍👨🏽 👩🏻‍❤️‍👨🏾 👩🏻‍❤️‍👨🏿 👩🏼‍❤️‍👨🏻 👩🏼‍❤️‍👨🏼 👩🏼‍❤️‍👨🏽 👩🏼‍❤️‍👨🏾 👩🏼‍❤️‍👨🏿 👩🏽‍❤️‍👨🏻 👩🏽‍❤️‍👨🏼 👩🏽‍❤️‍👨🏽 👩🏽‍❤️‍👨🏾 👩🏽‍❤️‍👨🏿 👩🏾‍❤️‍👨🏻 👩🏾‍❤️‍👨🏼 👩🏾‍❤️‍👨🏽"},{"id":"13","event":null,"data":"Сигурно некој мора да може да набави џем од боровинки? 👩🏾‍❤️‍👨🏾 👩🏾‍❤️‍👨🏿 👩🏿‍❤️‍👨🏻 👩🏿‍❤️‍👨🏼 👩🏿‍❤️‍👨🏽 👩🏿‍❤️‍👨🏾 👩🏿‍❤️‍👨🏿 👨‍❤️‍👨 👨🏻‍❤️‍👨🏻 👨🏻‍❤️‍👨🏼 👨🏻‍❤️‍👨🏽 👨🏻‍❤️‍👨🏾 👨🏻‍❤️‍👨🏿 👨🏼‍❤️‍👨🏻 👨🏼‍❤️‍👨🏼 👨🏼‍❤️‍👨🏽 👨🏼‍❤️‍👨🏾 👨🏼‍❤️‍👨🏿 👨🏽‍❤️‍👨🏻 👨🏽‍❤️‍👨🏼 👨🏽‍❤️‍👨🏽 👨🏽‍❤️‍👨🏾 👨🏽‍❤️‍👨🏿 👨🏾‍❤️‍👨🏻 👨🏾‍❤️‍👨🏼 👨🏾‍❤️‍👨🏽 👨🏾‍❤️‍👨🏾 👨🏾‍❤️‍👨🏿 👨🏿‍❤️‍👨🏻 👨🏿‍❤️‍👨🏼 👨🏿‍❤️‍👨🏽 👨🏿‍❤️‍👨🏾 👨🏿‍❤️‍👨🏿 👩‍❤️‍👩 👩🏻‍❤️‍👩🏻 👩🏻‍❤️‍👩🏼 👩🏻‍❤️‍👩🏽 👩🏻‍❤️‍👩🏾 👩🏻‍❤️‍👩🏿 👩🏼‍❤️‍👩🏻 👩🏼‍❤️‍👩🏼 👩🏼‍❤️‍👩🏽 👩🏼‍❤️‍👩🏾 👩🏼‍❤️‍👩🏿 👩🏽‍❤️‍👩🏻 👩🏽‍❤️‍👩🏼 👩🏽‍❤️‍👩🏽 👩🏽‍❤️‍👩🏾 👩🏽‍❤️‍👩🏿 👩🏾‍❤️‍👩🏻 👩🏾‍❤️‍👩🏼 👩🏾‍❤️‍👩🏽 👩🏾‍❤️‍👩🏾 👩🏾‍❤️‍👩🏿 👩🏿‍❤️‍👩🏻 👩🏿‍❤️‍👩🏼 👩🏿‍❤️‍👩🏽 👩🏿‍❤️‍👩🏾 👩🏿‍❤️‍👩🏿 👪 👨‍👩‍👦 👨‍👩‍👧 👨‍👩‍👧‍👦 👨‍👩‍👦‍👦 👨‍👩‍👧‍👧 👨‍👨‍👦 👨‍👨‍👧 👨‍👨‍👧‍👦 👨‍👨‍👦‍👦 👨‍👨‍👧‍👧 👩‍👩‍👦 👩‍👩‍👧 👩‍👩‍👧‍👦 👩‍👩‍👦‍👦 👩‍👩‍👧‍👧 👨‍👦 👨‍👦‍👦 👨‍👧 👨‍👧‍👦 👨‍👧‍👧 👩‍👦 👩‍👦‍👦 👩‍👧 👩‍👧‍👦 👩‍👧‍👧 🐕‍🦺 🐈‍⬛ 🐻‍❄️ 🏳️‍🌈 🏳️‍⚧️ 🏴‍☠️"},{"id":null,"event":"done","data":"✔"}]} \ No newline at end of file diff --git a/aimux-stream/tests/sse_test.rs b/aimux-stream/tests/sse_test.rs index 80f46365..cc1f595c 100644 --- a/aimux-stream/tests/sse_test.rs +++ b/aimux-stream/tests/sse_test.rs @@ -1,535 +1,860 @@ -//! Independent SSE parsing tests. +//! Port of the `eventsource-parser` test suite against [`aimux_stream::SseStream`]. //! -//! The TS SDK has no standalone SSE parser tests — `parseJsonEventStream` is -//! only exercised indirectly through provider tests. These tests fill that gap, -//! covering the edge cases listed in the task against [`aimux_stream::SseStream`]. +//! `eventsource-parser` is the parser behind the AI SDK's `parseJsonEventStream`. +//! This file ports `test/parse.test.ts` and `test/stream.test.ts` from v3.1.1 +//! (MIT), driven by the same fixtures (`test/fixtures.ts`; `test/multibyte.ts` +//! is checked in as `tests/fixtures/eventsource_parser_multibyte.json`). Test +//! names follow the upstream titles. //! -//! The expected behavior mirrors `eventsource-parser`'s `EventSourceParserStream` -//! (which the TS `parseJsonEventStream` pipes through): -//! - lines end in `\n`, `\r` or `\r\n` (freely mixed); an event is -//! dispatched only on a blank line; -//! - an event with no `data:` line is NOT dispatched (covers comment-only, -//! `event:`/`id:`/`retry:`-only, and blank-line keep-alives); -//! - exactly one leading U+0020 SPACE after the `:` is stripped from a field -//! value (per the SSE spec); -//! - a partial event at EOF (no terminating blank line) is dropped; -//! - a UTF-8 BOM at stream start is stripped; a line without `:` is a field -//! with an empty value; an empty `event` means no type; an `id` with -//! U+0000 is ignored; `retry` must be all ASCII digits; -//! - exceeding the buffer limit is fatal and ends the stream. +//! Adaptations forced by the Rust API (`SseStream` yields whole events, has no +//! callbacks, and works on bytes): +//! - `retry` is reported on the dispatched event instead of through `onRetry`, +//! so a `retry:` block with no `data:` is dropped, and upstream's +//! `reconnect-interval` events become `SseEvent::retry`. +//! - Upstream's `onError` for an invalid `retry` value has no counterpart: the +//! value is ignored, which is what is asserted. +//! - `maxBufferSize` maps to `with_max_event_size`; overflow yields +//! `SseError::FrameTooLarge` and ends the stream. +//! - A leading U+FEFF is the UTF-8 BOM at byte level. Upstream's +//! "invalid byte-order mark" case feeds a decoded U+FEFF that the JS +//! `TextDecoderStream` would already have stripped, so only the "multiple +//! places" case is ported. +//! +//! Not ported (no Rust counterpart): `onComment` call counts, `reset()` +//! (3 tests), `onError` `ParseError` payloads (3 tests), the "function passed +//! to `createParser`" guard, and the `onError: 'terminate'` variants of the +//! stream tests (covered by the single `FrameTooLarge` case). +//! +//! The `aimux` module holds the few tests for behaviour upstream does not have. use aimux_stream::{SseError, SseEvent, SseStream}; use bytes::Bytes; use futures::stream::{self, StreamExt}; +use serde_json::Value; +use sha2::{Digest, Sha256}; -/// Feed `chunks` (in arrival order) into a [`SseStream`] and collect all -/// emitted events. Each chunk simulates one `Result` item from a -/// real byte stream, so splitting a single event across chunks exercises the -/// cross-chunk buffering path. -async fn collect_events(chunks: Vec<&str>) -> Vec> { - let items: Vec> = chunks - .into_iter() - .map(|s| Ok(Bytes::copy_from_slice(s.as_bytes()))) - .collect(); - let stream = SseStream::new(stream::iter(items)); - stream.collect::>().await -} - -fn data(event: &SseEvent) -> &str { - &event.data -} +// ── harness ────────────────────────────────────────────────────────────── -// ── the 12 required edge cases ──────────────────────────────────────────── +type Item = Result; +type Triple<'a> = (Option<&'a str>, Option<&'a str>, &'a str); -#[tokio::test] -async fn single_complete_event() { - let events = collect_events(vec!["data: hello\n\n"]).await; - assert_eq!(events.len(), 1); - let event = events[0].as_ref().unwrap(); - assert_eq!(data(event), "hello"); - assert!(event.event.is_none()); - assert!(event.id.is_none()); - assert!(event.retry.is_none()); +async fn run_bytes(chunks: Vec>, max_event_size: Option) -> Vec { + let items: Vec> = + chunks.into_iter().map(|c| Ok(Bytes::from(c))).collect(); + match max_event_size { + Some(max) => { + SseStream::with_max_event_size(stream::iter(items), max) + .collect::>() + .await + } + None => { + SseStream::new(stream::iter(items)) + .collect::>() + .await + } + } } -#[tokio::test] -async fn multiple_consecutive_events() { - let events = collect_events(vec!["data: first\n\ndata: second\n\n"]).await; - assert_eq!(events.len(), 2); - assert_eq!(data(events[0].as_ref().unwrap()), "first"); - assert_eq!(data(events[1].as_ref().unwrap()), "second"); +async fn run(chunks: Vec, max_event_size: Option) -> Vec { + run_bytes( + chunks.into_iter().map(String::into_bytes).collect(), + max_event_size, + ) + .await } -#[tokio::test] -async fn event_spanning_two_chunks_half_line_split() { - // A single event whose data line arrives in two pieces. - let events = collect_events(vec!["data: hel", "lo\n\n"]).await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), "hello"); +/// Feed `chunks` and return the events; any error item fails the test. +async fn events(chunks: Vec) -> Vec { + run(chunks, None) + .await + .into_iter() + .map(|item| item.unwrap_or_else(|e| panic!("unexpected error item: {e:?}"))) + .collect() } -#[tokio::test] -async fn event_spanning_many_tiny_chunks() { - // The same event byte-split across many tiny chunks. - let events = collect_events(vec!["da", "ta: ", "wor", "ld", "\n", "\n"]).await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), "world"); +fn s(chunks: &[&str]) -> Vec { + chunks.iter().map(|c| (*c).to_string()).collect() } -#[tokio::test] -async fn multi_line_data_field() { - let events = collect_events(vec!["data: line1\ndata: line2\n\n"]).await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), "line1\nline2"); +/// `(id, event, data)` of each event. +fn triples(events: &[SseEvent]) -> Vec> { + events + .iter() + .map(|e| (e.id.as_deref(), e.event.as_deref(), e.data.as_str())) + .collect() } -#[tokio::test] -async fn event_field() { - let events = collect_events(vec!["event: message\ndata: payload\n\n"]).await; - assert_eq!(events.len(), 1); - let event = events[0].as_ref().unwrap(); - assert_eq!(event.event.as_deref(), Some("message")); - assert_eq!(data(event), "payload"); -} +// ── `eventsource-encoder` (used by upstream's fixtures) ────────────────── -#[tokio::test] -async fn id_field() { - let events = collect_events(vec!["id: 42\ndata: payload\n\n"]).await; - assert_eq!(events.len(), 1); - let event = events[0].as_ref().unwrap(); - assert_eq!(event.id.as_deref(), Some("42")); - assert_eq!(data(event), "payload"); +#[derive(Default)] +struct Msg<'a> { + event: Option<&'a str>, + retry: Option, + id: Option<&'a str>, + data: Option<&'a str>, } -#[tokio::test] -async fn retry_field() { - let events = collect_events(vec!["retry: 5000\ndata: payload\n\n"]).await; - assert_eq!(events.len(), 1); - let event = events[0].as_ref().unwrap(); - assert_eq!(event.retry, Some(5000)); - assert_eq!(data(event), "payload"); +fn encode_data(text: &str) -> String { + let normalized = text.replace("\r\n", "\n").replace('\r', "\n"); + let lines: Vec<&str> = normalized.split('\n').collect(); + let mut out = String::new(); + for (i, line) in lines.iter().enumerate() { + out.push_str("data: "); + out.push_str(line); + out.push_str(if i + 1 == lines.len() { "\n\n" } else { "\n" }); + } + out } -#[tokio::test] -async fn comment_lines_starting_with_colon_are_ignored() { - // A comment line within an event block does not affect the event. - let events = collect_events(vec![": this is a comment\ndata: hello\n\n"]).await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), "hello"); +fn encode_comment(comment: &str) -> String { + let normalized = comment.replace("\r\n", "\n").replace('\r', "\n"); + format!(": {}\n\n", normalized.replace('\n', "\n: ")) } -#[tokio::test] -async fn comment_only_event_is_not_emitted() { - // An event consisting solely of a comment has no data line → not dispatched. - let events = collect_events(vec![": just a comment\n\n"]).await; - assert!(events.is_empty()); +fn encode(msg: &Msg<'_>) -> String { + let mut out = String::new(); + if let Some(event) = msg.event.filter(|e| !e.is_empty()) { + out.push_str(&format!("event: {event}\n")); + } + if let Some(retry) = msg.retry { + out.push_str(&format!("retry: {retry}\n")); + } + if let Some(id) = msg.id { + out.push_str(&format!("id: {id}\n")); + } + if let Some(data) = msg.data { + out.push_str(&encode_data(data)); + } else if !out.is_empty() { + out.push_str("\n\n"); + } + out +} + +fn done() -> String { + encode(&Msg { + event: Some("done"), + data: Some("✔"), + ..Msg::default() + }) +} + +fn data_only(data: &str) -> String { + encode(&Msg { + data: Some(data), + ..Msg::default() + }) +} + +// ── fixtures (`test/fixtures.ts`) ──────────────────────────────────────── + +struct Multibyte { + lines: Vec, + emojis: Vec, + expected: Vec<(Option, Option, String)>, +} + +fn multibyte() -> Multibyte { + let root: Value = + serde_json::from_str(include_str!("fixtures/eventsource_parser_multibyte.json")).unwrap(); + let strings = |key: &str| -> Vec { + root[key] + .as_array() + .unwrap() + .iter() + .map(|v| v.as_str().unwrap().to_string()) + .collect() + }; + let optional = |v: &Value| v.as_str().map(str::to_string); + Multibyte { + lines: strings("lines"), + emojis: strings("emojis"), + expected: root["expected"] + .as_array() + .unwrap() + .iter() + .map(|e| { + ( + optional(&e["id"]), + optional(&e["event"]), + e["data"].as_str().unwrap().to_string(), + ) + }) + .collect(), + } } -#[tokio::test] -async fn done_sentinel_is_emitted_as_data() { - // The parser emits `data: [DONE]` like any other event; higher layers - // (e.g. the OpenAI provider) filter the sentinel. This mirrors the TS - // `parseJsonEventStream`, which drops `[DONE]` after parsing — the SSE - // parser itself does not special-case it. - let events = collect_events(vec!["data: [DONE]\n\n"]).await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), "[DONE]"); +/// Split `text` after `units` UTF-16 code units (JS `slice(0, n)` / `slice(n)`). +fn split_utf16(text: &str, units: usize) -> (&str, &str) { + let mut seen = 0; + for (index, ch) in text.char_indices() { + if seen == units { + return text.split_at(index); + } + seen += ch.len_utf16(); + assert!(seen <= units, "UTF-16 cut inside a surrogate pair"); + } + (text, "") +} + +fn multibyte_chunks(mb: &Multibyte) -> Vec { + let per_message = mb.emojis.len().div_ceil(mb.lines.len()); + let mut chunks = Vec::new(); + for (i, line) in mb.lines.iter().enumerate() { + let start = (per_message * i).min(mb.emojis.len()); + let end = (per_message * i + per_message).min(mb.emojis.len()); + let line = format!("{line} {}", mb.emojis[start..end].join(" ")); + chunks.push(format!("id: {i}\n")); + if i % 2 == 0 { + // Even lines are split into two `data:` lines. + let (head, tail) = split_utf16(&line, 5); + chunks.push(format!("data:{head}\n")); + chunks.push(format!("data:{tail}\n\n")); + } else { + chunks.push(format!("data:{line}\n\n")); + } + } + chunks.push(done()); + chunks } -#[tokio::test] -async fn blank_lines_and_heartbeat_emit_nothing() { - // Bare blank lines are keep-alives / heartbeats with no data → not emitted. - let events = collect_events(vec!["\n\n\n\n"]).await; - assert!(events.is_empty()); +fn iso_timestamp(i: usize) -> String { + format!("2026-09-29T05:51:{:02}.{:03}Z", i % 60, i) } -#[tokio::test] -async fn crlf_vs_lf_line_endings() { - // CRLF-terminated single event. - let events = collect_events(vec!["data: hello\r\n\r\n"]).await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), "hello"); +/// `^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}\.\d{3}Z$` +fn is_iso_timestamp(value: &str) -> bool { + const SHAPE: &[u8] = b"dddd-dd-ddTdd:dd:dd.dddZ"; + value.len() == SHAPE.len() + && value.bytes().zip(SHAPE).all(|(b, shape)| match shape { + b'd' => b.is_ascii_digit(), + other => b == *other, + }) } -#[tokio::test] -async fn crlf_multi_line_data() { - // CRLF-terminated multi-line data joins with a single `\n`. - let events = collect_events(vec!["data: line1\r\ndata: line2\r\n\r\n"]).await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), "line1\nline2"); -} +// ── parse.test.ts ──────────────────────────────────────────────────────── #[tokio::test] -async fn incomplete_event_at_eof_without_blank_line_is_dropped() { - // No terminating blank line → the buffered partial is not a complete event - // and is dropped (matches eventsource-parser's EventSourceParserStream, - // which has no flush handler). - let events = collect_events(vec!["data: hello"]).await; - assert!(events.is_empty()); +async fn basic_unnamed_events_stream() { + let mut chunks: Vec = (0..5).map(|i| data_only(&i.to_string())).collect(); + chunks.push(done()); + let got = events(chunks).await; + assert_eq!( + triples(&got), + vec![ + (None, None, "0"), + (None, None, "1"), + (None, None, "2"), + (None, None, "3"), + (None, None, "4"), + (None, Some("done"), "✔"), + ] + ); } #[tokio::test] -async fn incomplete_multi_line_event_at_eof_is_dropped() { - // A buffered data line that never receives its terminating blank line is - // dropped, even if the data line itself looks complete. - let events = collect_events(vec!["data: hello\n"]).await; - assert!(events.is_empty()); +async fn stream_of_time_event_name() { + let mut chunks: Vec = (0..5) + .map(|i| { + encode(&Msg { + event: Some("time"), + data: Some(&iso_timestamp(i)), + ..Msg::default() + }) + }) + .collect(); + chunks.push(done()); + let got = events(chunks).await; + assert_eq!(got.len(), 6); + for event in &got[..5] { + assert_eq!(event.event.as_deref(), Some("time")); + assert!(is_iso_timestamp(&event.data), "{}", event.data); + } } -// ── additional spec-faithfulness coverage ──────────────────────────────── - #[tokio::test] -async fn data_without_space_after_colon() { - // `data:hello` (no space) → value is `hello`. - let events = collect_events(vec!["data:hello\n\n"]).await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), "hello"); +async fn stream_of_time_event_names_unbalanced_chunks() { + let mut chunks = Vec::new(); + let mut ids = Vec::new(); + for i in 0..30 { + let id = (100_000_000_000_u64 + i as u64 * 7919).to_string(); + let message = encode(&Msg { + id: Some(&id), + event: Some("time"), + data: Some(&iso_timestamp(i)), + ..Msg::default() + }); + // Upstream splits at a random offset; the offset is varied + // deterministically here (the message is ASCII). + let split = (i * 11 + 3) % message.len(); + chunks.push(message[..split].to_string()); + chunks.push(message[split..].to_string()); + ids.push(id); + } + let got = events(chunks).await; + assert_eq!(got.len(), 30); + for (event, id) in got.iter().zip(&ids) { + assert_eq!(event.event.as_deref(), Some("time")); + assert_eq!(event.id.as_deref(), Some(id.as_str())); + assert!(is_iso_timestamp(&event.data), "{}", event.data); + } } #[tokio::test] -async fn only_one_leading_space_is_removed() { - // Per the SSE spec exactly ONE leading U+0020 SPACE is removed; the rest - // of the value is preserved verbatim. - let events = collect_events(vec!["data: two leading spaces\n\n"]).await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), " two leading spaces"); +async fn stream_of_identified_messages_and_retry_interval() { + let chunks = (1337..1339) + .map(|id| { + let id = id.to_string(); + encode(&Msg { + event: Some("tick"), + data: Some(&id), + id: Some(&id), + retry: Some(50), + }) + }) + .collect(); + let got = events(chunks).await; + assert_eq!( + triples(&got), + vec![ + (Some("1337"), Some("tick"), "1337"), + (Some("1338"), Some("tick"), "1338"), + ] + ); + assert!(got.iter().all(|e| e.retry == Some(50))); } #[tokio::test] -async fn empty_data_value() { - // `data:` with nothing after it contributes an empty data line. - let events = collect_events(vec!["data:\n\n"]).await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), ""); +async fn stream_of_heartbeat_comments_unnamed_events() { + let mut chunks = Vec::new(); + for letter in ['A', 'B', 'C', 'D', 'E'] { + chunks.push(encode_comment(" ♥")); + chunks.push(data_only(&letter.to_string())); + } + chunks.push(done()); + let got = events(chunks).await; + assert_eq!( + triples(&got), + vec![ + (None, None, "A"), + (None, None, "B"), + (None, None, "C"), + (None, None, "D"), + (None, None, "E"), + (None, Some("done"), "✔"), + ] + ); } #[tokio::test] -async fn event_field_without_data_is_not_emitted() { - // `event:` without a `data:` line → not dispatched. - let events = collect_events(vec!["event: ping\n\n"]).await; - assert!(events.is_empty()); +async fn stream_of_multi_line_data_events() { + let mut chunks = s(&[ + "event: stock\n", + "data: YHOO\n", + "data: +2\n", + "data: 10\n\n", + "event: stock\n", + "data: GOOG\n", + "data: -8\n", + "data: 1881\n\n", + ]); + chunks.push(done()); + let got = events(chunks).await; + assert_eq!( + triples(&got)[..2], + [ + (None, Some("stock"), "YHOO\n+2\n10"), + (None, Some("stock"), "GOOG\n-8\n1881"), + ] + ); } #[tokio::test] -async fn id_field_without_data_is_not_emitted() { - let events = collect_events(vec!["id: 99\n\n"]).await; - assert!(events.is_empty()); +async fn stream_of_multi_byte_events() { + let mb = multibyte(); + let got = events(multibyte_chunks(&mb)).await; + let got: Vec<_> = got.into_iter().map(|e| (e.id, e.event, e.data)).collect(); + assert_eq!(got, mb.expected); } #[tokio::test] -async fn retry_field_without_data_is_not_emitted() { - let events = collect_events(vec!["retry: 1000\n\n"]).await; - assert!(events.is_empty()); +async fn stream_of_multi_byte_events_with_some_empty_lines_thrown_in() { + let got = events(vec![ + "\n\n\n\nid: 1\ndata: 我現在都看實況不玩遊戲\n\n".to_string(), + done(), + ]) + .await; + assert_eq!( + triples(&got), + vec![ + (Some("1"), None, "我現在都看實況不玩遊戲"), + (None, Some("done"), "✔"), + ] + ); } #[tokio::test] -async fn event_with_all_fields_together() { - let events = collect_events(vec!["event: update\nid: 7\nretry: 3000\ndata: payload\n\n"]).await; - assert_eq!(events.len(), 1); - let e = events[0].as_ref().unwrap(); - assert_eq!(e.event.as_deref(), Some("update")); - assert_eq!(e.id.as_deref(), Some("7")); - assert_eq!(e.retry, Some(3000)); - assert_eq!(data(e), "payload"); +async fn stream_of_leading_bom() { + let got = events(vec![ + "\u{FEFF}data: bomful 1\n\n".to_string(), + "data: bomless 2\n\n".to_string(), + done(), + ]) + .await; + assert_eq!( + triples(&got), + vec![ + (None, None, "bomful 1"), + (None, None, "bomless 2"), + (None, Some("done"), "✔"), + ] + ); } #[tokio::test] -async fn intermixed_comments_between_events() { - let events = collect_events(vec![ - ": heartbeat\ndata: first\n\n", - ": another comment\ndata: second\n\n", +async fn stream_containing_byte_order_mark_multiple_places() { + let got = events(vec![ + "\u{FEFF}data: bomful 1\n\n".to_string(), + "\u{FEFF}data: bomful 2\n\n".to_string(), + "data: bomless 3\n\n".to_string(), + done(), ]) .await; - assert_eq!(events.len(), 2); - assert_eq!(data(events[0].as_ref().unwrap()), "first"); - assert_eq!(data(events[1].as_ref().unwrap()), "second"); + // Only the BOM at the start of the stream is stripped; the second one + // turns the field name into an unknown field, so that event has no data. + assert_eq!( + triples(&got), + vec![ + (None, None, "bomful 1"), + (None, None, "bomless 3"), + (None, Some("done"), "✔"), + ] + ); } #[tokio::test] -async fn invalid_retry_value_is_ignored() { - // A non-numeric `retry:` value is ignored; the event is still dispatched - // because it has a data line. - let events = collect_events(vec!["retry: not-a-number\ndata: payload\n\n"]).await; - assert_eq!(events.len(), 1); - let e = events[0].as_ref().unwrap(); - assert_eq!(e.retry, None); - assert_eq!(data(e), "payload"); +async fn stream_using_carriage_returns() { + let mut chunks = s(&[ + "data: dog\r", + "data: bark\r\r", + "data: cat\r", + "data: meow\r\r", + ]); + chunks.push(done()); + let got = events(chunks).await; + assert_eq!( + triples(&got), + vec![ + (None, None, "dog\nbark"), + (None, None, "cat\nmeow"), + (None, Some("done"), "✔"), + ] + ); } #[tokio::test] -async fn unknown_field_is_ignored() { - let events = collect_events(vec!["foo: bar\ndata: payload\n\n"]).await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), "payload"); +async fn stream_using_line_feeds() { + let mut chunks = s(&[ + "data: cow\n", + "data: moo\n\n", + "data: horse\n", + "data: neigh\n\n", + ]); + chunks.push(done()); + let got = events(chunks).await; + assert_eq!( + triples(&got), + vec![ + (None, None, "cow\nmoo"), + (None, None, "horse\nneigh"), + (None, Some("done"), "✔"), + ] + ); } #[tokio::test] -async fn mixed_lf_and_crlf_events() { - // An LF event followed by a CRLF event in the same stream. - let events = collect_events(vec!["data: lf\n\ndata: crlf\r\n\r\n"]).await; - assert_eq!(events.len(), 2); - assert_eq!(data(events[0].as_ref().unwrap()), "lf"); - assert_eq!(data(events[1].as_ref().unwrap()), "crlf"); -} - -// ── cross-chunk UTF-8 reassembly (P0-03) ───────────────────────────────── - -/// Like [`collect_events`] but takes raw byte chunks, so a chunk that ends in -/// the middle of a multi-byte UTF-8 sequence (not representable as a `&str`) -/// can be fed in. -async fn collect_events_bytes(chunks: Vec>) -> Vec> { - let items: Vec> = - chunks.into_iter().map(|c| Ok(Bytes::from(c))).collect(); - let stream = SseStream::new(stream::iter(items)); - stream.collect::>().await +async fn stream_using_carriage_returns_and_line_feeds() { + let mut chunks = s(&[ + "data: sheep\r\n", + "data: bleat\r\n\r\n", + "data: pig\r\n", + "data: oink\r\n\r\n", + ]); + chunks.push(done()); + let got = events(chunks).await; + assert_eq!( + triples(&got), + vec![ + (None, None, "sheep\nbleat"), + (None, None, "pig\noink"), + (None, Some("done"), "✔"), + ] + ); } #[tokio::test] -async fn chinese_sse_event_split_at_every_byte_boundary() { - // A Chinese SSE event whose data contains multi-byte UTF-8. Splitting the - // full event at *every* byte position must still reassemble to the intact - // text. The old per-chunk `String::from_utf8_lossy` decoder would corrupt - // a code point split across two chunks into two U+FFFDs. - let payload = "data: 你好世界\n\n"; - let bytes = payload.as_bytes(); - for split in 1..bytes.len() { - let (a, b) = bytes.split_at(split); - let events = collect_events_bytes(vec![a.to_vec(), b.to_vec()]).await; - assert_eq!( - events.len(), - 1, - "split at byte {} produced {} results", - split, - events.len() - ); - let event = events[0] - .as_ref() - .unwrap_or_else(|e| panic!("split at byte {split} errored: {e:?}")); - assert_eq!( - event.data, "你好世界", - "split at byte {split} corrupted the data" - ); - } +async fn stream_with_varying_odd_uses_of_comments() { + let mb = multibyte(); + let mut chunks = s(&[": Hello\n\n"]); + chunks.push(":".repeat(300)); + chunks.extend(s(&[ + "\n", + "data: First\n\n", + ": Первый", + ": 第二", + "\n", + "data: Second\n\n", + ])); + chunks.extend(std::iter::repeat_n(": Moop \n".to_string(), 10)); + chunks.extend(s(&[ + ": ثالث", + "\n", + "data: Third\n\n", + ":നാലാമത്തെ", + "\n", + "data: Fourth\n\n", + ])); + chunks.push(format!(": {} :", mb.emojis[..100].join(" "))); + chunks.extend(s(&["\n", "data: Fifth\n\n"])); + chunks.push(done()); + let got = events(chunks).await; + assert_eq!( + triples(&got), + vec![ + (None, None, "First"), + (None, None, "Second"), + (None, None, "Third"), + (None, None, "Fourth"), + (None, None, "Fifth"), + (None, Some("done"), "✔"), + ] + ); } #[tokio::test] -async fn invalid_utf8_frame_returns_utf8_error() { - // `data: ` followed by 0xE4 0xBD — the first two bytes of `你` (U+4F60 = - // E4 BD A0) without the trailing byte — is an incomplete UTF-8 sequence. - // Strict decoding of the complete frame surfaces a `Utf8` error instead - // of silently producing a replacement char. - let chunk = vec![b'd', b'a', b't', b'a', b':', b' ', 0xE4, 0xBD, b'\n', b'\n']; - let events = collect_events_bytes(vec![chunk]).await; - assert_eq!(events.len(), 1, "expected exactly one result"); - match &events[0] { - Err(SseError::Utf8(_)) => {} - other => panic!("expected SseError::Utf8, got {other:?}"), - } +async fn stream_with_even_more_odd_uses_of_comments() { + let long = "x".repeat(2 * 1024 + 1); + let chunks = vec![ + "data:1\r\r:\0\n:\r\ndata:2\n\n:".to_string(), + long.clone(), + "\rdata:3\n\n:data:fail\r:".to_string(), + long, + "\ndata:4\n\n".to_string(), + "data:5".to_string(), + ]; + let got = events(chunks).await; + // No newline after the last message, thus not emitted. + assert_eq!( + triples(&got), + vec![ + (None, None, "1"), + (None, None, "2"), + (None, None, "3"), + (None, None, "4"), + ] + ); } -// ── bounded buffers (P1-10) ───────────────────────────────────────────── - #[tokio::test] -async fn oversized_frame_returns_frame_too_large() { - // A complete event whose frame (`data: hello world`) exceeds the 10-byte - // limit is rejected; the frame+terminator is dropped so the stream ends. - let stream = SseStream::with_max_event_size( - stream::iter(vec![Ok::<_, std::io::Error>(Bytes::from_static( - b"data: hello world\n\n", - ))]), - 10, +async fn stream_with_empty_event_field() { + let got = events(vec![ + "event:\ndata: Hello 1\n\n".to_string(), + "event:\n\n".to_string(), + done(), + ]) + .await; + assert_eq!( + triples(&got), + vec![(None, None, "Hello 1"), (None, Some("done"), "✔")] ); - let results: Vec<_> = stream.collect().await; - assert_eq!(results.len(), 1); - assert!(matches!(results[0], Err(SseError::FrameTooLarge))); } #[tokio::test] -async fn buffer_growing_past_limit_without_terminator_returns_frame_too_large() { - // No terminator ever arrives, so the buffer would grow unboundedly; the - // limit trips instead (previously this allocated forever). - let stream = SseStream::with_max_event_size( - stream::iter(vec![Ok::<_, std::io::Error>(Bytes::from_static( - b"data: no terminator here", - ))]), - 10, +async fn stream_with_empty_retry_field() { + let got = events(vec![ + encode(&Msg { + id: Some("1"), + retry: Some(500), + data: Some("🥌"), + ..Msg::default() + }), + "id:2\nretry:\ndata:🧹\n\n".to_string(), + encode(&Msg { + id: Some("3"), + data: Some("✅"), + ..Msg::default() + }), + ]) + .await; + assert_eq!( + triples(&got), + vec![ + (Some("1"), None, "🥌"), + (Some("2"), None, "🧹"), + (Some("3"), None, "✅"), + ] + ); + // The empty `retry` is ignored; `retry: 500` belongs to the first event. + assert_eq!( + got.iter().map(|e| e.retry).collect::>(), + vec![Some(500), None, None] ); - let results: Vec<_> = stream.collect().await; - assert_eq!(results.len(), 1); - assert!(matches!(results[0], Err(SseError::FrameTooLarge))); } -// ── eventsource-parser alignment ───────────────────────────────────────── - #[tokio::test] -async fn bare_cr_line_endings() { - let events = collect_events(vec![ - "event: a\rdata: one\rdata: two\r\rdata: next\r\r:keep-alive\n", +async fn stream_with_oddly_shaped_data_field() { + let got = events(vec![ + "data:\n\ndata\ndata\n\ndata:test\n\n".to_string(), + done(), ]) .await; - assert_eq!(events.len(), 2); - let first = events[0].as_ref().unwrap(); - assert_eq!(first.event.as_deref(), Some("a")); - assert_eq!(data(first), "one\ntwo"); - assert_eq!(data(events[1].as_ref().unwrap()), "next"); + // `data:\n\n` dispatches an event with empty data; `data\ndata\n\n` is two + // empty data lines, joined by a newline. + assert_eq!( + triples(&got), + vec![ + (None, None, ""), + (None, None, "\n"), + (None, None, "test"), + (None, Some("done"), "✔"), + ] + ); } #[tokio::test] -async fn trailing_cr_at_end_of_stream_does_not_dispatch() { - // A `\r` at the end of the buffered input waits for a possible `\n`; with - // no flush at end of stream (as `EventSourceParserStream`) the event is - // never completed. - let events = collect_events(vec!["data: a\r\rdata: b\r\r"]).await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), "a"); +async fn stream_with_cr_separating_chunks_of_same_event() { + // A CR at the end of a chunk may be half of a CRLF, so it must not end + // the line yet: otherwise `A\nB` and `C` would be two events. + // https://github.com/rexxars/eventsource-parser/issues/17 + let got = events(s(&["data: A\r\n", "data: B\r", "\n", "data: C\r\n", "\n"])).await; + assert_eq!(triples(&got), vec![(None, None, "A\nB\nC")]); } #[tokio::test] -async fn mixed_terminators_within_one_event() { - // `\r\n` then a bare `\n` blank line: one event, as eventsource-parser. - let events = collect_events(vec!["data: a\r\ndata: b\n\ndata: c\r\r\n"]).await; - assert_eq!(events.len(), 2); - assert_eq!(data(events[0].as_ref().unwrap()), "a\nb"); - assert_eq!(data(events[1].as_ref().unwrap()), "c"); +async fn stream_with_partially_incorrect_retry_fields() { + let got = events(s(&["retry:1000\nretry:2000x\ndata:x\n\n"])).await; + // `2000x` is not all ASCII digits and is ignored; `1000` is kept. + assert_eq!(triples(&got), vec![(None, None, "x")]); + assert_eq!(got[0].retry, Some(1000)); } #[tokio::test] -async fn crlf_split_between_chunks_is_one_terminator() { - // A `\r` at the end of a chunk must wait: the `\n` in the next chunk - // completes the same terminator, not an extra blank line. - let events = collect_events(vec!["data: a\r", "\ndata: b\r", "\n\r", "\n"]).await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), "a\nb"); +async fn stream_with_id_field_containing_a_null_character() { + // An `id` containing U+0000 is ignored, so the earlier `123` survives. + let got = events(s(&["id: 123\nid: bad\0id\ndata: hello\n\n"])).await; + assert_eq!(triples(&got), vec![(Some("123"), None, "hello")]); } #[tokio::test] -async fn leading_bom_is_stripped() { - let events = collect_events(vec!["\u{FEFF}data: first\n\n"]).await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), "first"); +async fn stream_with_incorrect_retry_fields() { + let got = events(s(&[ + "\nretry: 500\n\ndata: first\n\nretry: 50x\n\ndata: second\n\n", + ])) + .await; + assert_eq!( + triples(&got), + vec![(None, None, "first"), (None, None, "second")] + ); } #[tokio::test] -async fn leading_bom_split_across_chunks_is_stripped() { - let bom = "\u{FEFF}".as_bytes(); - let events = collect_events_bytes(vec![ - bom[..1].to_vec(), - bom[1..].to_vec(), - b"data: first\n\n".to_vec(), +async fn stream_with_unknown_fields_in_the_stream() { + let got = events(vec![ + "data:abc\n data\ndata\nfoobar:xxx\njustsometext\n:thisisacommentyay\ndata:123\n\n" + .to_string(), + done(), ]) .await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), "first"); + assert_eq!( + triples(&got), + vec![(None, None, "abc\n\n123"), (None, Some("done"), "✔")] + ); } #[tokio::test] -async fn bom_after_stream_start_is_not_stripped() { - // Only the very start of the stream may carry a BOM; later it is part of - // the field name, which is then unknown. - let events = collect_events(vec!["data: a\n\n\u{FEFF}data: b\n\n"]).await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), "a"); +async fn stream_with_huge_data_chunks() { + const TEN_MEGABYTES: usize = 1024 * 1024 * 10; + const EXPECTED_SHA256: &str = + "e094a44a2436226ea9feb04e413a28de012b406012ec0eb6b37ad0a19d403660"; + + let mb = multibyte(); + let data_chunk = encode_data(&format!( + "{}\n{}", + mb.lines.join("\n\n"), + mb.emojis.join(" ") + )) + .trim() + .to_string(); + let mut chunks = s(&[": hello\n\n"]); + let mut written = 0; + while written < TEN_MEGABYTES { + chunks.push(data_chunk.clone()); + written += data_chunk.len(); + } + chunks.extend(s(&["\n\n", ": END-OF-STREAM\n\n"])); + chunks.push(encode(&Msg { + event: Some("done"), + data: Some(EXPECTED_SHA256), + ..Msg::default() + })); + + // The default limit (1 MiB) would end the stream; upstream is unbounded. + let got: Vec = run(chunks, Some(usize::MAX)) + .await + .into_iter() + .map(Result::unwrap) + .collect(); + assert_eq!(got.len(), 2); + // JS `String.length` counts UTF-16 code units. + assert_eq!(got[0].data.encode_utf16().count(), 4_808_512); + assert_eq!( + format!("{:x}", Sha256::digest(got[0].data.as_bytes())), + got[1].data + ); } #[tokio::test] -async fn field_without_colon_has_empty_value() { - // `data` alone is a data line with an empty value. - let events = collect_events(vec!["data\n\ndata\ndata: x\n\n"]).await; - assert_eq!(events.len(), 2); - assert_eq!(data(events[0].as_ref().unwrap()), ""); - assert_eq!(data(events[1].as_ref().unwrap()), "\nx"); +async fn max_buffer_size_triggers_on_pending_fragment_overflow_no_terminator() { + // Under the limit: nothing yet. + assert!(run(s(&["short start"]), Some(16)).await.is_empty()); + // The accumulated fragments pass the limit. + let results = run(s(&["short start", " and now too long"]), Some(16)).await; + assert_eq!(results.len(), 1); + assert!(matches!(results[0], Err(SseError::FrameTooLarge))); } #[tokio::test] -async fn empty_event_field_means_no_event_type() { - let events = collect_events(vec!["event:\ndata: x\n\n"]).await; - assert_eq!(events[0].as_ref().unwrap().event, None); +async fn max_buffer_size_triggers_on_data_buffer_overflow_no_blank_line() { + // Each `data:` line grows the event's data; without a blank line it + // accumulates until the limit trips, and no event is dispatched. + let chunks = (0..50).map(|i| format!("data: chunk-{i}\n")).collect(); + let results = run(chunks, Some(32)).await; + assert_eq!(results.len(), 1); + assert!(matches!(results[0], Err(SseError::FrameTooLarge))); } #[tokio::test] -async fn event_type_resets_after_each_dispatch() { - let events = collect_events(vec!["event: a\ndata: 1\n\ndata: 2\n\n"]).await; - assert_eq!(events[0].as_ref().unwrap().event.as_deref(), Some("a")); - assert_eq!(events[1].as_ref().unwrap().event, None); +async fn max_buffer_size_ends_the_stream_after_overflow() { + // Upstream: `feed` throws after an overflow until `reset()`. Here the + // stream ends, so later events are never yielded. + let results = run( + s(&[ + "this is too long for the buffer", + "data: hello\n\n", + "data: world\n\n", + ]), + Some(8), + ) + .await; + assert_eq!(results.len(), 1); + assert!(matches!(results[0], Err(SseError::FrameTooLarge))); } #[tokio::test] -async fn id_containing_null_is_ignored() { - let events = collect_events(vec!["id: ok\nid: bad\u{0}id\ndata: x\n\n"]).await; - assert_eq!(events[0].as_ref().unwrap().id.as_deref(), Some("ok")); +async fn max_buffer_size_not_triggered_when_events_dispatch_within_the_limit() { + let chunks = (0..100).map(|i| format!("data: {i}\n\n")).collect(); + let got: Vec = run(chunks, Some(64)) + .await + .into_iter() + .map(Result::unwrap) + .collect(); + assert_eq!(got.len(), 100); } #[tokio::test] -async fn retry_must_be_all_ascii_digits() { - // `retry: 5` loses its one leading space and is valid; a second space is not. - for value in ["+5", "-5", "5s", " 5", ""] { - let input = format!("retry:{value}\ndata: x\n\n"); - let events = collect_events(vec![input.as_str()]).await; - assert_eq!(events[0].as_ref().unwrap().retry, None, "retry {value:?}"); - } - let events = collect_events(vec!["retry:1500\ndata: x\n\n"]).await; - assert_eq!(events[0].as_ref().unwrap().retry, Some(1500)); +async fn max_buffer_size_large_input_within_the_default_limit_is_not_an_error() { + // Upstream's default is unbounded; aimux's is 1 MiB (see `aimux` below). + assert!(run(vec!["x".repeat(1_000_000)], None).await.is_empty()); } -#[tokio::test] -async fn field_names_are_case_sensitive() { - let events = collect_events(vec!["Data: x\n\ndata: y\n\n"]).await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), "y"); -} +// ── stream.test.ts ─────────────────────────────────────────────────────── #[tokio::test] -async fn invalid_utf8_drops_only_the_current_event() { - let mut bytes = b"data: ".to_vec(); - bytes.extend_from_slice(&[0xE4, 0xBD]); - bytes.extend_from_slice(b"\ndata: tail\n\ndata: next\n\n"); - let events = collect_events_bytes(vec![bytes]).await; - // The rest of the poisoned event (`tail`) is discarded, not dispatched as - // a fragment; the following event is unaffected. - assert_eq!(events.len(), 2); - assert!(matches!(events[0], Err(SseError::Utf8(_)))); - assert_eq!(data(events[1].as_ref().unwrap()), "next"); +async fn can_use_event_source_parser_stream() { + let chunks = (0..10) + .map(|i| { + encode(&Msg { + event: Some("foo"), + id: Some(&format!("evt-{i}")), + data: Some(&format!("Hello {i}")), + ..Msg::default() + }) + }) + .collect::>() + .concat(); + let got = events(vec![chunks]).await; + assert_eq!(got.len(), 10); + assert_eq!(triples(&got)[0], (Some("evt-0"), Some("foo"), "Hello 0")); + assert_eq!(triples(&got)[9], (Some("evt-9"), Some("foo"), "Hello 9")); } #[tokio::test] -async fn exceeding_the_limit_ends_the_stream() { - let stream = SseStream::with_max_event_size( - stream::iter(vec![ - Ok::<_, std::io::Error>(Bytes::from_static(b"data: ok\n\n")), - Ok(Bytes::from_static(b"data: far too long for the limit\n\n")), - Ok(Bytes::from_static(b"data: ok\n\n")), - ]), - 10, - ); - let results: Vec<_> = stream.collect().await; - assert_eq!(results.len(), 2); - assert_eq!(data(results[0].as_ref().unwrap()), "ok"); - assert!(matches!(results[1], Err(SseError::FrameTooLarge))); +async fn max_buffer_size_terminates_the_stream() { + // Upstream also runs this with `onError: 'terminate'` and with a custom + // `onError`; overflow is fatal either way. + let results = run(vec!["x".repeat(1024)], Some(64)).await; + assert_eq!(results.len(), 1); + assert!(matches!(results[0], Err(SseError::FrameTooLarge))); } -#[tokio::test] -async fn multi_event_payload_split_at_every_byte_boundary() { - let payload = "event: e\r\nid: 1\r\ndata: 你好\r\n\r\ndata: a\rdata: b\r\r:c\ndata: z\n\n"; - let bytes = payload.as_bytes(); - let expected = collect_events(vec![payload]).await; - let expected: Vec<(Option, String, Option)> = expected - .into_iter() - .map(|e| { - let e = e.unwrap(); - (e.event, e.data, e.id) - }) - .collect(); - assert_eq!(expected.len(), 3); - for split in 1..bytes.len() { - let (a, b) = bytes.split_at(split); - let got: Vec<_> = collect_events_bytes(vec![a.to_vec(), b.to_vec()]) - .await - .into_iter() - .map(|e| { - let e = e.unwrap_or_else(|err| panic!("split {split}: {err:?}")); - (e.event, e.data, e.id) - }) - .collect(); - assert_eq!(got, expected, "split at byte {split}"); +// ── aimux-specific behaviour (not covered upstream) ────────────────────── + +mod aimux { + use super::*; + + #[tokio::test] + async fn invalid_utf8_discards_the_rest_of_the_event_only() { + // Upstream decodes lossily (`TextDecoder`); SseStream decodes strictly. + // `E4 BD` is the start of `你` without its last byte. + let mut bytes = b"data: ".to_vec(); + bytes.extend_from_slice(&[0xE4, 0xBD]); + bytes.extend_from_slice(b"\ndata: tail\n\ndata: next\n\n"); + let results = run_bytes(vec![bytes], None).await; + assert_eq!(results.len(), 2); + assert!(matches!(results[0], Err(SseError::Utf8(_)))); + assert_eq!(results[1].as_ref().unwrap().data, "next"); + } + + #[tokio::test] + async fn multibyte_fixture_split_at_arbitrary_byte_sizes() { + // Chunks can end inside a code point or inside the BOM; the parser + // works on bytes, so the result must not depend on where they end. + let mb = multibyte(); + let mut stream_bytes = "\u{FEFF}".as_bytes().to_vec(); + stream_bytes.extend(multibyte_chunks(&mb).concat().into_bytes()); + for size in [1, 2, 3, 5, 8, 13] { + let chunks = stream_bytes.chunks(size).map(<[u8]>::to_vec).collect(); + let got: Vec<_> = run_bytes(chunks, None) + .await + .into_iter() + .map(|item| { + let e = item.unwrap_or_else(|err| panic!("chunk size {size}: {err:?}")); + (e.id, e.event, e.data) + }) + .collect(); + assert_eq!(got, mb.expected, "chunk size {size}"); + } + } + + #[tokio::test] + async fn transport_error_is_reported_as_a_stream_error() { + let items: Vec> = vec![ + Ok(Bytes::from_static(b"data: a\n\n")), + Err(std::io::Error::other("boom")), + Ok(Bytes::from_static(b"data: b\n\n")), + ]; + let results = SseStream::new(stream::iter(items)) + .collect::>() + .await; + assert_eq!(results.len(), 3); + assert_eq!(results[0].as_ref().unwrap().data, "a"); + assert!(matches!(&results[1], Err(SseError::Stream(m)) if m == "boom")); + assert_eq!(results[2].as_ref().unwrap().data, "b"); + } + + #[tokio::test] + async fn default_limit_is_one_mebibyte() { + let results = run(vec!["x".repeat(1024 * 1024 + 1)], None).await; + assert_eq!(results.len(), 1); + assert!(matches!(results[0], Err(SseError::FrameTooLarge))); } } diff --git a/aimux-stream/tests/streaming_tool_call_argument_state_test.rs b/aimux-stream/tests/streaming_tool_call_argument_state_test.rs index 44cf754e..4d7fba63 100644 --- a/aimux-stream/tests/streaming_tool_call_argument_state_test.rs +++ b/aimux-stream/tests/streaming_tool_call_argument_state_test.rs @@ -12,6 +12,7 @@ fn starts_with_structured_value_true() { #[test] fn starts_with_structured_value_false() { + // TS covers `undefined` and `null` separately; Rust has one `None`. assert!(!starts_with_structured_value(None)); for value in ["", " ", "1", "\"value\""] { assert!(!starts_with_structured_value(Some(value)), "{value:?}"); From 20b2b502d10ae27b0af665aa5b7fc6bd1daeea83 Mon Sep 17 00:00:00 2001 From: cunninghamcard-bit Date: Thu, 1 Oct 2026 15:44:11 +0000 Subject: [PATCH 8/9] feat(stream)!: decode SSE with the sse-stream crate Replace aimux-stream's hand-written SSE parser with a thin adapter over sse-stream 0.3 (WHATWG event-stream parsing), the way the AI SDK's parseJsonEventStream wraps eventsource-parser. The adapter keeps the AI SDK dispatch rules: only blocks with a data line are dispatched, an empty `event:` is None, and `retry` rides on its event. - No event size limit any more, as upstream: `with_max_event_size` and `SseError::FrameTooLarge` are removed. Real recordings of OpenAI Responses and Gemini image generation carry 2-3 MB events; the 1 MiB cap dropped them, and after the previous commit ended the stream on the first one. Both recordings now decode completely (15/15 and 5/5 events). - `SseError::Stream` carries the transport error as its source instead of a String; `Utf8` holds a `str::Utf8Error`; `Decode` covers the rest. Decoder errors end the stream after being reported. - The ported eventsource-parser suite runs unchanged against the adapter (29 cases); the 6 `maxBufferSize` cases go with the limit. - provider-utils: the `SseError::Stream` arm formats the source error. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01WGwanvo9WLU9HWD7sRR8TS --- CHANGELOG.md | 24 +- aimux-provider-utils/src/response_handler.rs | 4 +- aimux-stream/Cargo.toml | 1 + aimux-stream/src/sse.rs | 311 ++++--------------- aimux-stream/tests/sse_test.rs | 130 ++------ 5 files changed, 111 insertions(+), 359 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index dce2b580..74fd598e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -24,15 +24,21 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 be `Send + Sync`. - Added `StreamingToolCallArgumentState` and `starts_with_structured_value`, the structural JSON-prefix tracker the new correlation logic relies on. -- `SseStream` rewritten as a line-based parser matching `eventsource-parser` - (the parser behind the AI SDK's `parseJsonEventStream`): `\n`, `\r` and - `\r\n` line endings in any mix, a leading UTF-8 BOM is stripped, a field - line without `:` counts as an empty value, an empty `event:` is `None`, an - `id` containing U+0000 is ignored, and `retry` must be all ASCII digits. - Exceeding `max_event_size` (now: buffered event data plus the partial line) - yields `SseError::FrameTooLarge` and ends the stream instead of skipping the - frame. Invalid UTF-8 still yields one `SseError::Utf8` and discards that - event; later events are unaffected. +- `SseStream` is now a thin adapter over the `sse-stream` crate (WHATWG + event-stream parsing), mirroring how the AI SDK's `parseJsonEventStream` + wraps `eventsource-parser`: `\n`, `\r` and `\r\n` line endings in any + mix, a leading UTF-8 BOM is stripped, a field line without `:` counts as an + empty value, an empty `event:` is `None`, an `id` containing U+0000 is + ignored, `retry` must be all ASCII digits, and a block is dispatched only + when it had a `data` line. There is no event size limit any more (as + upstream): `with_max_event_size` and `SseError::FrameTooLarge` are gone, + which fixes streams carrying multi-megabyte events (OpenAI Responses and + Gemini image generation). `SseError::Stream` carries the transport error + as its source instead of a `String`; `SseError::Utf8` holds a + `str::Utf8Error`; `SseError::Decode` is added. Decoder errors (invalid + UTF-8, transport failure) now end the stream after being reported. + `SseStream` requires the body error type to implement `std::error::Error + + Send + Sync + 'static`. - Removed `NdjsonStream` / `NdjsonError`: nothing in the workspace used them and the AI SDK has no counterpart. `tokio` is now a dev-dependency only and the unused direct `serde` dependency is dropped. diff --git a/aimux-provider-utils/src/response_handler.rs b/aimux-provider-utils/src/response_handler.rs index db568ef8..2042669d 100644 --- a/aimux-provider-utils/src/response_handler.rs +++ b/aimux-provider-utils/src/response_handler.rs @@ -392,14 +392,14 @@ where match item { Ok(event) if event.data == "[DONE]" => continue, Ok(event) => yield serde_json::from_str::(&event.data).map_err(AiMuxError::from), - Err(aimux_stream::SseError::Stream(message)) => { + Err(aimux_stream::SseError::Stream(error)) => { // Preserve response transport failures as ApiCallError // items. Framing/parser failures remain JsonParse below. yield Err(AiMuxError::ApiCall(Box::new(ApiCallError { response_headers: Some(stream_headers.clone()), is_retryable: true, ..ApiCallError::new( - message, + error.to_string(), stream_url.clone(), stream_request_body_values.clone(), ) diff --git a/aimux-stream/Cargo.toml b/aimux-stream/Cargo.toml index 2d51d4d6..85b6bb20 100644 --- a/aimux-stream/Cargo.toml +++ b/aimux-stream/Cargo.toml @@ -14,6 +14,7 @@ pin-project-lite = { workspace = true } bytes = { workspace = true } serde_json = { workspace = true } thiserror = { workspace = true } +sse-stream = { version = "0.3.0", default-features = false, features = ["memchr"] } [dev-dependencies] sha2 = "0.10" diff --git a/aimux-stream/src/sse.rs b/aimux-stream/src/sse.rs index a9be7a66..f32f436c 100644 --- a/aimux-stream/src/sse.rs +++ b/aimux-stream/src/sse.rs @@ -1,49 +1,60 @@ -//! SSE (Server-Sent Events) parser for streaming model responses. +//! SSE (Server-Sent Events) decoding for streaming model responses. //! -//! Line-based incremental parser following `eventsource-parser`'s -//! `createParser` (the parser behind the AI SDK's `parseJsonEventStream`) and -//! the WHATWG "parsing an event stream" algorithm: +//! The parser is the [`sse-stream`](https://crates.io/crates/sse-stream) crate +//! (the WHATWG "parsing an event stream" algorithm: `\n`, `\r` and `\r\n` +//! line endings in any mix, a leading UTF-8 BOM, `field: value` with one +//! optional leading space, lines without `:` as empty-valued fields, +//! `:`-prefixed comments, an `id` containing U+0000 ignored, `retry` only when +//! all ASCII digits, unknown fields ignored, a partial block at end of stream +//! dropped). //! -//! - lines end in `\n`, `\r` or `\r\n`, freely mixed; a `\r` at the end of -//! the buffered input waits for the next chunk in case a `\n` follows; -//! - a UTF-8 BOM at the very start of the stream is stripped; -//! - a blank line dispatches the event, but only if it had at least one -//! `data` line; field state resets after every blank line; -//! - `field: value` removes exactly one leading U+0020 SPACE; a line without -//! `:` is a field with an empty value; `:`-prefixed lines are comments; -//! - an empty `event` value means no event type; an `id` containing U+0000 is -//! ignored; `retry` is accepted only when it is all ASCII digits; unknown -//! fields are ignored; -//! - a partial event at end of stream is not dispatched. +//! This module is the thin adapter that gives it the AI SDK's dispatch +//! semantics, the way `parseJsonEventStream` wraps `eventsource-parser`: //! -//! Two aimux additions: every line is strictly UTF-8 decoded after it has -//! been reassembled (a code point split across chunks is never corrupted, and -//! invalid UTF-8 surfaces as [`SseError::Utf8`] and discards the rest of -//! that event up to the next blank line), and buffered input is bounded by `max_event_size` (the parser's -//! `maxBufferSize`; exceeding it is fatal and ends the stream, as upstream). +//! - a block is dispatched only if it had at least one `data` line; comment- +//! only and metadata-only blocks are dropped; +//! - an empty `event` value means no event type; +//! - `retry` is reported on the event it was parsed with (upstream reports it +//! through `onRetry`), so a `retry` block without `data` is dropped; +//! - there is no buffer size limit, as in `parseJsonEventStream`. +//! +//! Differences from upstream kept on purpose: a field value that is not valid +//! UTF-8 is a strict [`SseError::Utf8`] and ends the stream (upstream decodes +//! lossily), and a transport error ends the stream after it is reported. -use std::collections::VecDeque; +use std::marker::PhantomData; +use std::pin::Pin; +use std::task::{Context, Poll, ready}; use bytes::Bytes; use futures::Stream; use pin_project_lite::pin_project; -use std::pin::Pin; -use std::task::{Context, Poll}; +use sse_stream::SseByteStream; use thiserror::Error; -/// Default upper bound on buffered event data plus any partial line (1 MiB). -const DEFAULT_MAX_EVENT_SIZE: usize = 1024 * 1024; - -const BOM: &[u8] = b"\xEF\xBB\xBF"; - +/// A terminal decoding error: after yielding one, the stream ends. #[derive(Debug, Error)] pub enum SseError { + /// A recognized field value (`data`, `event`, `id`, `retry`) is not valid + /// UTF-8. #[error("utf-8 decode error: {0}")] - Utf8(#[from] std::string::FromUtf8Error), + Utf8(#[source] std::str::Utf8Error), + /// The underlying byte stream failed; the source is the transport error. #[error("stream error: {0}")] - Stream(String), - #[error("SSE frame exceeded maximum allowed size")] - FrameTooLarge, + Stream(#[source] Box), + /// Any other decoder error. + #[error("SSE decode error: {0}")] + Decode(#[source] sse_stream::Error), +} + +impl From for SseError { + fn from(error: sse_stream::Error) -> Self { + match error { + sse_stream::Error::Body(source) => Self::Stream(source), + sse_stream::Error::Utf8Parse(source) => Self::Utf8(source), + other => Self::Decode(other), + } + } } /// A parsed SSE event. @@ -59,238 +70,54 @@ pub struct SseEvent { pub retry: Option, } -/// Incremental SSE line parser (the `createParser` state machine). -#[derive(Debug)] -struct Parser { - /// Bytes after the last complete line terminator. - pending: Vec, - /// Offset in `pending` from which the next terminator search starts, so a - /// long line fed in many small chunks is scanned once, not quadratically. - scan_from: usize, - bom_checked: bool, - event: Option, - data: String, - data_lines: usize, - id: Option, - retry: Option, - /// Set after a line fails UTF-8 decoding: the rest of the event (up to - /// the next blank line) is discarded rather than dispatched as a fragment. - poisoned: bool, - max_size: usize, - terminated: bool, - ready: VecDeque>, -} - -impl Parser { - fn new(max_size: usize) -> Self { - Self { - pending: Vec::new(), - scan_from: 0, - bom_checked: false, - event: None, - data: String::new(), - data_lines: 0, - id: None, - retry: None, - poisoned: false, - max_size, - terminated: false, - ready: VecDeque::new(), - } - } - - fn feed(&mut self, chunk: &[u8]) { - if self.terminated { - return; - } - self.pending.extend_from_slice(chunk); - - if !self.bom_checked { - // Wait until the first three bytes can rule a BOM in or out. - if self.pending.len() < BOM.len() && BOM.starts_with(&self.pending) { - return; - } - if self.pending.starts_with(BOM) { - self.pending.drain(..BOM.len()); - } - self.bom_checked = true; - } - - let mut line_start = 0; - let mut search = self.scan_from; - while let Some(offset) = self.pending[search..] - .iter() - .position(|&b| b == b'\n' || b == b'\r') - { - let end = search + offset; - let is_cr = self.pending[end] == b'\r'; - // A trailing `\r` may be the first half of a `\r\n` split across - // chunks: defer it until more input arrives. - if is_cr && end + 1 == self.pending.len() { - break; - } - let line = self.pending[line_start..end].to_vec(); - self.parse_line(&line); - if self.terminated { - return; - } - line_start = end + 1; - if is_cr && self.pending.get(line_start) == Some(&b'\n') { - line_start += 1; - } - search = line_start; - } - self.pending.drain(..line_start); - // Resume the next search at the deferred `\r`, if any. - self.scan_from = self.pending.len().saturating_sub(1); - if self.pending.last() != Some(&b'\r') { - self.scan_from = self.pending.len(); - } - self.check_size(self.pending.len()); - } - - fn parse_line(&mut self, line: &[u8]) { - if line.is_empty() { - self.dispatch(); - return; - } - if self.poisoned { - return; - } - let line = match String::from_utf8(line.to_vec()) { - Ok(line) => line, - Err(error) => { - // The event being built can no longer be trusted: drop it and - // everything up to the next blank line. - self.reset_event(); - self.poisoned = true; - self.ready.push_back(Err(SseError::Utf8(error))); - return; - } - }; - if line.starts_with(':') { - return; // comment - } - let (field, value) = match line.split_once(':') { - Some((field, value)) => (field, value.strip_prefix(' ').unwrap_or(value)), - None => (line.as_str(), ""), - }; - match field { - "event" => self.event = (!value.is_empty()).then(|| value.to_string()), - "data" => { - if self.data_lines > 0 { - self.data.push('\n'); - } - self.data.push_str(value); - self.data_lines += 1; - // Lines already consumed from `pending` are not buffered any - // more; only the event data counts until the chunk is done. - self.check_size(0); - } - "id" => { - if !value.contains('\0') { - self.id = Some(value.to_string()); - } - } - "retry" if !value.is_empty() && value.bytes().all(|b| b.is_ascii_digit()) => { - self.retry = value.parse().ok(); - } - _ => {} // unknown field - } - } - - fn dispatch(&mut self) { - if self.data_lines > 0 { - self.ready.push_back(Ok(SseEvent { - event: self.event.take(), - data: std::mem::take(&mut self.data), - id: self.id.take(), - retry: self.retry.take(), - })); - } - self.reset_event(); - } - - fn reset_event(&mut self) { - self.poisoned = false; - self.event = None; - self.data.clear(); - self.data_lines = 0; - self.id = None; - self.retry = None; - } - - /// `maxBufferSize`: buffered event data plus the partial line. Exceeding - /// it is fatal, as in `EventSourceParserStream`. - fn check_size(&mut self, pending_len: usize) { - if self.terminated || pending_len + self.data.len() <= self.max_size { - return; - } - self.terminated = true; - self.pending.clear(); - self.reset_event(); - self.ready.push_back(Err(SseError::FrameTooLarge)); - } -} - pin_project! { /// An adapter that decodes a byte stream into SSE events. - pub struct SseStream { + pub struct SseStream + where + S: Stream>, + { #[pin] - inner: S, - parser: Parser, - done: bool, - _err: std::marker::PhantomData, + inner: SseByteStream, + _err: PhantomData, } } impl SseStream where - S: Stream> + Unpin, + S: Stream>, { pub fn new(stream: S) -> Self { - Self::with_max_event_size(stream, DEFAULT_MAX_EVENT_SIZE) - } - - /// Create an [`SseStream`] with a custom buffer limit: when buffered event - /// data plus the current partial line exceed `max_event_size` bytes, the - /// stream yields [`SseError::FrameTooLarge`] and ends. - pub fn with_max_event_size(stream: S, max_event_size: usize) -> Self { Self { - inner: stream, - parser: Parser::new(max_event_size), - done: false, - _err: std::marker::PhantomData, + inner: SseByteStream::new(stream), + _err: PhantomData, } } } impl Stream for SseStream where - S: Stream> + Unpin, - E: std::fmt::Display, + S: Stream>, + E: std::error::Error + Send + Sync + 'static, { type Item = Result; - fn poll_next(mut self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll> { - let this = self.as_mut().get_mut(); - + fn poll_next(self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll> { + let mut this = self.project(); loop { - if let Some(item) = this.parser.ready.pop_front() { - return Poll::Ready(Some(item)); - } - // A partial event at end of stream is not dispatched: the - // upstream stream has no flush handler. - if this.done || this.parser.terminated { - return Poll::Ready(None); - } - match Pin::new(&mut this.inner).poll_next(cx) { - Poll::Ready(Some(Ok(bytes))) => this.parser.feed(&bytes), - Poll::Ready(Some(Err(e))) => { - return Poll::Ready(Some(Err(SseError::Stream(e.to_string())))); + match ready!(this.inner.as_mut().poll_next(cx)) { + None => return Poll::Ready(None), + Some(Err(error)) => return Poll::Ready(Some(Err(error.into()))), + Some(Ok(block)) => { + // Dispatch only blocks that carried at least one `data` + // line (`dataLines > 0` upstream). + let Some(data) = block.data else { continue }; + return Poll::Ready(Some(Ok(SseEvent { + event: block.event.filter(|event| !event.is_empty()), + data, + id: block.id, + retry: block.retry, + }))); } - Poll::Ready(None) => this.done = true, - Poll::Pending => return Poll::Pending, } } } diff --git a/aimux-stream/tests/sse_test.rs b/aimux-stream/tests/sse_test.rs index cc1f595c..2672a962 100644 --- a/aimux-stream/tests/sse_test.rs +++ b/aimux-stream/tests/sse_test.rs @@ -13,8 +13,8 @@ //! `reconnect-interval` events become `SseEvent::retry`. //! - Upstream's `onError` for an invalid `retry` value has no counterpart: the //! value is ignored, which is what is asserted. -//! - `maxBufferSize` maps to `with_max_event_size`; overflow yields -//! `SseError::FrameTooLarge` and ends the stream. +//! - There is no `maxBufferSize`: like `parseJsonEventStream`, the stream is +//! unbounded, so the upstream overflow cases are not ported. //! - A leading U+FEFF is the UTF-8 BOM at byte level. Upstream's //! "invalid byte-order mark" case feeds a decoded U+FEFF that the JS //! `TextDecoderStream` would already have stripped, so only the "multiple @@ -22,8 +22,8 @@ //! //! Not ported (no Rust counterpart): `onComment` call counts, `reset()` //! (3 tests), `onError` `ParseError` payloads (3 tests), the "function passed -//! to `createParser`" guard, and the `onError: 'terminate'` variants of the -//! stream tests (covered by the single `FrameTooLarge` case). +//! to `createParser`" guard, and the `maxBufferSize` / `onError: 'terminate'` +//! stream tests (no size limit here). //! //! The `aimux` module holds the few tests for behaviour upstream does not have. @@ -38,34 +38,21 @@ use sha2::{Digest, Sha256}; type Item = Result; type Triple<'a> = (Option<&'a str>, Option<&'a str>, &'a str); -async fn run_bytes(chunks: Vec>, max_event_size: Option) -> Vec { +async fn run_bytes(chunks: Vec>) -> Vec { let items: Vec> = chunks.into_iter().map(|c| Ok(Bytes::from(c))).collect(); - match max_event_size { - Some(max) => { - SseStream::with_max_event_size(stream::iter(items), max) - .collect::>() - .await - } - None => { - SseStream::new(stream::iter(items)) - .collect::>() - .await - } - } + SseStream::new(stream::iter(items)) + .collect::>() + .await } -async fn run(chunks: Vec, max_event_size: Option) -> Vec { - run_bytes( - chunks.into_iter().map(String::into_bytes).collect(), - max_event_size, - ) - .await +async fn run(chunks: Vec) -> Vec { + run_bytes(chunks.into_iter().map(String::into_bytes).collect()).await } /// Feed `chunks` and return the events; any error item fails the test. async fn events(chunks: Vec) -> Vec { - run(chunks, None) + run(chunks) .await .into_iter() .map(|item| item.unwrap_or_else(|e| panic!("unexpected error item: {e:?}"))) @@ -697,12 +684,7 @@ async fn stream_with_huge_data_chunks() { ..Msg::default() })); - // The default limit (1 MiB) would end the stream; upstream is unbounded. - let got: Vec = run(chunks, Some(usize::MAX)) - .await - .into_iter() - .map(Result::unwrap) - .collect(); + let got: Vec = run(chunks).await.into_iter().map(Result::unwrap).collect(); assert_eq!(got.len(), 2); // JS `String.length` counts UTF-16 code units. assert_eq!(got[0].data.encode_utf16().count(), 4_808_512); @@ -713,57 +695,9 @@ async fn stream_with_huge_data_chunks() { } #[tokio::test] -async fn max_buffer_size_triggers_on_pending_fragment_overflow_no_terminator() { - // Under the limit: nothing yet. - assert!(run(s(&["short start"]), Some(16)).await.is_empty()); - // The accumulated fragments pass the limit. - let results = run(s(&["short start", " and now too long"]), Some(16)).await; - assert_eq!(results.len(), 1); - assert!(matches!(results[0], Err(SseError::FrameTooLarge))); -} - -#[tokio::test] -async fn max_buffer_size_triggers_on_data_buffer_overflow_no_blank_line() { - // Each `data:` line grows the event's data; without a blank line it - // accumulates until the limit trips, and no event is dispatched. - let chunks = (0..50).map(|i| format!("data: chunk-{i}\n")).collect(); - let results = run(chunks, Some(32)).await; - assert_eq!(results.len(), 1); - assert!(matches!(results[0], Err(SseError::FrameTooLarge))); -} - -#[tokio::test] -async fn max_buffer_size_ends_the_stream_after_overflow() { - // Upstream: `feed` throws after an overflow until `reset()`. Here the - // stream ends, so later events are never yielded. - let results = run( - s(&[ - "this is too long for the buffer", - "data: hello\n\n", - "data: world\n\n", - ]), - Some(8), - ) - .await; - assert_eq!(results.len(), 1); - assert!(matches!(results[0], Err(SseError::FrameTooLarge))); -} - -#[tokio::test] -async fn max_buffer_size_not_triggered_when_events_dispatch_within_the_limit() { - let chunks = (0..100).map(|i| format!("data: {i}\n\n")).collect(); - let got: Vec = run(chunks, Some(64)) - .await - .into_iter() - .map(Result::unwrap) - .collect(); - assert_eq!(got.len(), 100); -} - -#[tokio::test] -async fn max_buffer_size_large_input_within_the_default_limit_is_not_an_error() { - // Upstream's default is unbounded; aimux's is 1 MiB (see `aimux` below). - assert!(run(vec!["x".repeat(1_000_000)], None).await.is_empty()); +async fn large_input_without_a_terminator_is_not_an_error() { + // Unbounded, as upstream: a partial block is simply never dispatched. + assert!(run(vec!["x".repeat(4 * 1024 * 1024)]).await.is_empty()); } // ── stream.test.ts ─────────────────────────────────────────────────────── @@ -787,31 +721,22 @@ async fn can_use_event_source_parser_stream() { assert_eq!(triples(&got)[9], (Some("evt-9"), Some("foo"), "Hello 9")); } -#[tokio::test] -async fn max_buffer_size_terminates_the_stream() { - // Upstream also runs this with `onError: 'terminate'` and with a custom - // `onError`; overflow is fatal either way. - let results = run(vec!["x".repeat(1024)], Some(64)).await; - assert_eq!(results.len(), 1); - assert!(matches!(results[0], Err(SseError::FrameTooLarge))); -} - // ── aimux-specific behaviour (not covered upstream) ────────────────────── mod aimux { use super::*; #[tokio::test] - async fn invalid_utf8_discards_the_rest_of_the_event_only() { - // Upstream decodes lossily (`TextDecoder`); SseStream decodes strictly. + async fn invalid_utf8_is_a_terminal_error() { + // Upstream decodes lossily (`TextDecoder`); SseStream decodes strictly + // and, like every decoder error, ends the stream after reporting it. // `E4 BD` is the start of `你` without its last byte. let mut bytes = b"data: ".to_vec(); bytes.extend_from_slice(&[0xE4, 0xBD]); bytes.extend_from_slice(b"\ndata: tail\n\ndata: next\n\n"); - let results = run_bytes(vec![bytes], None).await; - assert_eq!(results.len(), 2); + let results = run_bytes(vec![bytes]).await; + assert_eq!(results.len(), 1); assert!(matches!(results[0], Err(SseError::Utf8(_)))); - assert_eq!(results[1].as_ref().unwrap().data, "next"); } #[tokio::test] @@ -823,7 +748,7 @@ mod aimux { stream_bytes.extend(multibyte_chunks(&mb).concat().into_bytes()); for size in [1, 2, 3, 5, 8, 13] { let chunks = stream_bytes.chunks(size).map(<[u8]>::to_vec).collect(); - let got: Vec<_> = run_bytes(chunks, None) + let got: Vec<_> = run_bytes(chunks) .await .into_iter() .map(|item| { @@ -845,16 +770,9 @@ mod aimux { let results = SseStream::new(stream::iter(items)) .collect::>() .await; - assert_eq!(results.len(), 3); + // The transport error is terminal: nothing after it is read. + assert_eq!(results.len(), 2); assert_eq!(results[0].as_ref().unwrap().data, "a"); - assert!(matches!(&results[1], Err(SseError::Stream(m)) if m == "boom")); - assert_eq!(results[2].as_ref().unwrap().data, "b"); - } - - #[tokio::test] - async fn default_limit_is_one_mebibyte() { - let results = run(vec!["x".repeat(1024 * 1024 + 1)], None).await; - assert_eq!(results.len(), 1); - assert!(matches!(results[0], Err(SseError::FrameTooLarge))); + assert!(matches!(&results[1], Err(SseError::Stream(e)) if e.to_string() == "boom")); } } From ff2517c57ab50828ffd1fa41bcc9f067678c1b3f Mon Sep 17 00:00:00 2001 From: cunninghamcard-bit Date: Thu, 1 Oct 2026 16:09:31 +0000 Subject: [PATCH 9/9] feat(provider-utils)!: move the tool-call tracker next to its users and wire the OpenAI stream to it The AI SDK keeps StreamingToolCallTracker in @ai-sdk/provider-utils, where every provider that speaks the OpenAI chat-completions wire format imports it. aimux had the tracker in aimux-stream (the eventsource-parser role), with its own ToolCallStreamPart event type and no callers, while openai/model.rs kept an index-only accumulator. This commit puts the pieces where the AI SDK has them. aimux-stream - StreamingToolCallTracker, StreamingToolCallArgumentState, starts_with_structured_value and ToolCallStreamPart move out; the crate is SSE decoding only. serde_json becomes a dev-dependency (test fixture); the `tool-calls` keyword is dropped. aimux-provider-utils - streaming_tool_call_tracker.rs / streaming_tool_call_argument_state.rs land here, logic unchanged. The tracker emits aimux_core::StreamPart ToolInputStart / ToolInputDelta / ToolInputEnd / ToolCall directly, so there is no second event type (RFC-0036: ToolCallStreamPart must not be a second public protocol). The generic metadata parameter becomes serde_json::Value / ProviderMetadata; TrackerError converts into AiMuxError::InvalidResponseData. The ported upstream tests (45 + 9) move with it, projecting StreamPart back to the upstream event shape. aimux-providers - openai/model.rs replaces its HashMap with the tracker: tool_calls deltas are correlated by wire id, index and function name instead of index alone; id-less continuations follow their call; indices reused across parallel calls stay distinct; ambiguous deltas are dropped; a call without a wire id gets a generated `tool-call` / `tool-call-N` id instead of an empty string; a new call without a function name ends the stream with InvalidResponseData (AI SDK behaviour) instead of starting a call with an empty name. - DeltaToolCall.index is Option (the AI SDK schema is `index: z.number().nullish()`). Every registry-backed provider goes through openai/model.rs, so this is the one plug point until S4-2 folds the mistral and xai chat copies into the same implementation. Verification: cargo test -p aimux-stream -p aimux-provider-utils; cargo test -p aimux-providers (129 suites, 3041 passed, 0 failed, including the cassette replays); cargo clippy -p aimux-stream -p aimux-provider-utils -p aimux-providers --all-targets -- -D warnings; cargo fmt --all -- --check. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01WGwanvo9WLU9HWD7sRR8TS --- CHANGELOG.md | 26 ++++ CONTRIBUTING.md | 4 +- README.md | 8 +- aimux-provider-utils/Cargo.toml | 1 + aimux-provider-utils/src/lib.rs | 12 +- .../src/streaming_tool_call_argument_state.rs | 0 .../src/streaming_tool_call_tracker.rs | 119 +++++++++-------- ...streaming_tool_call_argument_state_test.rs | 2 +- .../tests/streaming_tool_call_tracker_test.rs | 124 +++++++++++++----- aimux-providers/src/openai/model.rs | 109 +++++---------- aimux-providers/src/openai/types.rs | 4 +- aimux-stream/Cargo.toml | 4 +- aimux-stream/src/lib.rs | 16 +-- docs/PROJECT-OVERVIEW.md | 4 +- 14 files changed, 244 insertions(+), 189 deletions(-) rename {aimux-stream => aimux-provider-utils}/src/streaming_tool_call_argument_state.rs (100%) rename {aimux-stream => aimux-provider-utils}/src/streaming_tool_call_tracker.rs (87%) rename {aimux-stream => aimux-provider-utils}/tests/streaming_tool_call_argument_state_test.rs (96%) rename {aimux-stream => aimux-provider-utils}/tests/streaming_tool_call_tracker_test.rs (91%) diff --git a/CHANGELOG.md b/CHANGELOG.md index 74fd598e..b66e23f8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -42,6 +42,32 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - Removed `NdjsonStream` / `NdjsonError`: nothing in the workspace used them and the AI SDK has no counterpart. `tokio` is now a dev-dependency only and the unused direct `serde` dependency is dropped. +- `StreamingToolCallTracker`, `StreamingToolCallArgumentState`, + `starts_with_structured_value` and `ToolCallStreamPart` are no longer in + `aimux-stream`, which is now SSE decoding only (the role + `eventsource-parser` plays for the AI SDK). + +**Rust (aimux-provider-utils)** + +- Gains `StreamingToolCallTracker` (plus `StreamingToolCallDelta`, + `StreamingToolCallFunction`, `TypeValidation`, `TrackerError`, + `StreamingToolCallArgumentState`), where `@ai-sdk/provider-utils` keeps it. + It emits `aimux_core::StreamPart` tool-input parts directly; the separate + `ToolCallStreamPart` event type is gone. `TrackerError` converts into + `AiMuxError::InvalidResponseData`. Metadata hooks are typed with + `serde_json::Value` / `ProviderMetadata` instead of a generic parameter. + +**Rust (aimux-providers)** + +- The OpenAI chat-completions stream (`openai/model.rs`, which serves every + registry-backed provider) correlates `tool_calls` deltas with the tracker + instead of by `index` alone: deltas are matched by wire id, index and + function name; a continuation without an id follows its call; indices + reused across parallel calls stay distinct; ambiguous deltas are dropped; + a call whose delta carries no id gets a generated `tool-call` / + `tool-call-N` id instead of an empty string; a new call without a function + name ends the stream with `InvalidResponseData` (previously it started a + call with an empty name). `DeltaToolCall.index` is now `Option`. ## [0.5.0] - 2026-09-27 diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index ab11942a..2b302fb8 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -9,8 +9,8 @@ to set up a development environment, run the tests, and submit changes. aimux/ ├── aimux-core/ # Core abstractions: LanguageModel / Provider / Message / StreamPart ├── aimux-providers/ # provider implementations + cassettes (counts: docs/api/providers.md) -├── aimux-stream/ # SSE parsing, streamed tool-call tracking -├── aimux-provider-utils/ # HTTP utilities: retry, backoff, error parsing, API-key loading +├── aimux-stream/ # SSE decoding +├── aimux-provider-utils/ # HTTP utilities: retry, backoff, error parsing, API-key loading, streamed tool-call tracking ├── aimux-ffi/ # C ABI (opaque handle + JSON + push callback) for non-native bindings ├── bindings/ # Node, Python, Swift, Kotlin, Flutter, Go, C — share one Rust core ├── contract-tests/ # Shared JSON fixtures exercised across languages diff --git a/README.md b/README.md index 02ee0ed0..71fcfe0b 100644 --- a/README.md +++ b/README.md @@ -98,8 +98,8 @@ middleware, and telemetry per request). aimux/ ├── aimux-core # Core abstractions: LanguageModel / Provider / Message / StreamPart ├── aimux-providers # Provider implementations — registry-backed + typed (docs/api/providers.md) -├── aimux-stream # SSE parsing, streamed tool-call tracking -├── aimux-provider-utils # One-exchange HTTP helpers, response handlers, API-key loading +├── aimux-stream # SSE decoding +├── aimux-provider-utils # One-exchange HTTP helpers, response handlers, API-key loading, streamed tool-call tracking ├── aimux-ffi # C ABI (opaque handles + JSON results + owned aimux_error_t *) for non-native bindings └── tools/ # aimux-cli (cache probe) · aimux-replay · aimux-web (console) ``` @@ -122,8 +122,8 @@ cargo add aimux-core aimux-providers |-------|-------------|-----------| | `aimux-core` | Core abstractions: `LanguageModel` / `Provider` / `Message` / `StreamPart` | [crates.io](https://crates.io/crates/aimux-core) | | `aimux-providers` | Provider implementations — [registry-backed + typed](docs/api/providers.md) | [crates.io](https://crates.io/crates/aimux-providers) | -| `aimux-stream` | SSE parsing, streamed tool-call tracking | [crates.io](https://crates.io/crates/aimux-stream) | -| `aimux-provider-utils` | One-exchange HTTP helpers and typed response handlers | [crates.io](https://crates.io/crates/aimux-provider-utils) | +| `aimux-stream` | SSE decoding | [crates.io](https://crates.io/crates/aimux-stream) | +| `aimux-provider-utils` | One-exchange HTTP helpers, typed response handlers, streamed tool-call tracking | [crates.io](https://crates.io/crates/aimux-provider-utils) | | `aimux-ffi` | C ABI for non-native bindings | [crates.io](https://crates.io/crates/aimux-ffi) | **Node.js**: diff --git a/aimux-provider-utils/Cargo.toml b/aimux-provider-utils/Cargo.toml index 29a6049f..bf427a09 100644 --- a/aimux-provider-utils/Cargo.toml +++ b/aimux-provider-utils/Cargo.toml @@ -16,6 +16,7 @@ reqwest = { workspace = true } http = "1" serde = { workspace = true } serde_json = { workspace = true } +thiserror = { workspace = true } tokio = { workspace = true } tracing = { workspace = true } tracing-subscriber = { workspace = true } diff --git a/aimux-provider-utils/src/lib.rs b/aimux-provider-utils/src/lib.rs index 679f24be..fa33eb5d 100644 --- a/aimux-provider-utils/src/lib.rs +++ b/aimux-provider-utils/src/lib.rs @@ -3,7 +3,8 @@ //! Shared utilities for provider implementations. //! //! Provides one-exchange HTTP helpers, response handlers, API key loading, -//! header management, and URL utilities — the Rust equivalents of +//! header management, URL utilities and the streamed tool-call tracker for +//! the OpenAI chat-completions wire format — the Rust equivalents of //! `@ai-sdk/provider-utils`. Operation retry and timeout live in `aimux-core`. pub mod api_key; @@ -19,6 +20,8 @@ pub mod post_to_api; pub mod read_response_with_size_limit; pub mod response_handler; pub mod retry; +pub mod streaming_tool_call_argument_state; +pub mod streaming_tool_call_tracker; pub mod url; /// WebSocket client for realtime provider APIs (RFC-0028). Empty unless the /// `ws` feature is enabled. @@ -43,4 +46,11 @@ pub use response_handler::{ stream_error_api_call, }; pub use retry::RetryConfig; +pub use streaming_tool_call_argument_state::{ + StreamingToolCallArgumentState, starts_with_structured_value, +}; +pub use streaming_tool_call_tracker::{ + StreamingToolCallDelta, StreamingToolCallFunction, StreamingToolCallTracker, TrackerError, + TypeValidation, +}; pub use url::{validate_base_url, without_trailing_slash, without_trailing_slash_opt}; diff --git a/aimux-stream/src/streaming_tool_call_argument_state.rs b/aimux-provider-utils/src/streaming_tool_call_argument_state.rs similarity index 100% rename from aimux-stream/src/streaming_tool_call_argument_state.rs rename to aimux-provider-utils/src/streaming_tool_call_argument_state.rs diff --git a/aimux-stream/src/streaming_tool_call_tracker.rs b/aimux-provider-utils/src/streaming_tool_call_tracker.rs similarity index 87% rename from aimux-stream/src/streaming_tool_call_tracker.rs rename to aimux-provider-utils/src/streaming_tool_call_tracker.rs index 9a8d7240..ce9666dd 100644 --- a/aimux-stream/src/streaming_tool_call_tracker.rs +++ b/aimux-provider-utils/src/streaming_tool_call_tracker.rs @@ -5,8 +5,9 @@ //! //! Tracks streaming tool call state across the deltas of an OpenAI-compatible //! chat completion stream: accumulates `arguments` fragments, emits -//! `tool-input-start` / `tool-input-delta` / `tool-input-end` / `tool-call` -//! parts, and finalizes unfinished calls on [`StreamingToolCallTracker::flush`]. +//! [`StreamPart::ToolInputStart`] / [`StreamPart::ToolInputDelta`] / +//! [`StreamPart::ToolInputEnd`] / [`StreamPart::ToolCall`] parts, and +//! finalizes unfinished calls on [`StreamingToolCallTracker::flush`]. //! //! Deltas are correlated to calls by wire `id`, `index` and function name //! (see [`StreamingToolCallTracker`]'s resolution table), not by `index` @@ -16,12 +17,22 @@ //! Like the TS original, a call is *never* finalized before `flush`: a //! parsable argument buffer can still be the prefix of a longer argument //! string, so acting on it early would use truncated inputs (ai-sdk #13137). +//! +//! This is a tool for the OpenAI chat-completions wire format only. Protocols +//! whose streams carry explicit tool-call boundaries (Anthropic content +//! blocks, Google complete `functionCall` parts, Bedrock content blocks, +//! Cohere `tool-call-*` events, the Responses API's output items) do not need +//! it. use std::collections::{HashMap, HashSet}; use serde_json::Value; use thiserror::Error; +use aimux_core::error::AiMuxError; +use aimux_core::stream_part::StreamPart; +use aimux_core::types::ProviderMetadata; + use crate::streaming_tool_call_argument_state::{ StreamingToolCallArgumentState, starts_with_structured_value, }; @@ -98,28 +109,6 @@ impl StreamingToolCallDelta { } } -/// The stream parts emitted by [`StreamingToolCallTracker`]. -/// -/// Mirrors the subset of `LanguageModelV4StreamPart` the TS tracker enqueues. -/// [`ToolCallStreamPart::ToolCall::input`] is the raw accumulated argument -/// *string*; the tracker does not parse it. -#[derive(Debug, Clone, PartialEq)] -pub enum ToolCallStreamPart { - /// Start of a tool call's input streaming. - ToolInputStart { id: String, tool_name: String }, - /// A partial argument fragment. - ToolInputDelta { id: String, delta: String }, - /// End of a tool call's input streaming. - ToolInputEnd { id: String }, - /// A complete, finalized tool call. - ToolCall { - tool_call_id: String, - tool_name: String, - input: String, - provider_metadata: Option, - }, -} - /// How to validate the `type` field on a new tool call delta. #[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] pub enum TypeValidation { @@ -133,6 +122,9 @@ pub enum TypeValidation { } /// Errors raised while processing a tool call delta. +/// +/// The TS tracker throws `InvalidResponseDataError`; these convert into +/// [`AiMuxError::InvalidResponseData`]. #[derive(Debug, Error, PartialEq, Eq)] pub enum TrackerError { #[error("Expected 'function.name' to be a string.")] @@ -143,7 +135,13 @@ pub enum TrackerError { IdExhausted, } -struct TrackedToolCall { +impl From for AiMuxError { + fn from(error: TrackerError) -> Self { + AiMuxError::InvalidResponseData(error.to_string()) + } +} + +struct TrackedToolCall { id: String, index: Option, sequence: usize, @@ -151,7 +149,7 @@ struct TrackedToolCall { arguments: String, argument_state: StreamingToolCallArgumentState, has_finished: bool, - metadata: Option, + metadata: Option, } enum ToolCallResolution { @@ -161,17 +159,14 @@ enum ToolCallResolution { } type GenerateIdFn = Box String + Send + Sync>; -type ExtractMetadataFn = Box Option + Send + Sync>; -type BuildMetadataFn = Box) -> Option + Send + Sync>; +type ExtractMetadataFn = Box Option + Send + Sync>; +type BuildMetadataFn = Box) -> Option + Send + Sync>; /// Tracks streaming tool call state across multiple deltas. /// /// [`process_delta`](Self::process_delta) and [`flush`](Self::flush) return -/// the parts to forward downstream (the TS tracker enqueues them on a -/// controller instead). -/// -/// The type parameter `M` is the provider-metadata type; use `()` (the -/// default) when no metadata handling is needed. +/// the [`StreamPart`]s to forward downstream (the TS tracker enqueues them on +/// a controller instead). /// /// # Correlation /// @@ -186,25 +181,25 @@ type BuildMetadataFn = Box) -> Option + Send + Sync>; /// | absent | absent | unnamed | sole unfinished call, new call, or ambiguity | /// /// An *ambiguous* delta is dropped. -pub struct StreamingToolCallTracker { - tool_calls: Vec>, +pub struct StreamingToolCallTracker { + tool_calls: Vec, tool_calls_by_id: HashMap>, tool_calls_by_index: HashMap>, used_tool_call_ids: HashSet, next_generated_id_suffixes: HashMap, generate_id: GenerateIdFn, type_validation: TypeValidation, - extract_metadata: Option>, - build_provider_metadata: Option>, + extract_metadata: Option, + build_provider_metadata: Option, } -impl Default for StreamingToolCallTracker { +impl Default for StreamingToolCallTracker { fn default() -> Self { Self::new() } } -impl StreamingToolCallTracker { +impl StreamingToolCallTracker { /// Create a tracker with no metadata handling and default settings. /// /// The default id generator returns a blank-free constant, so ids that @@ -240,10 +235,11 @@ impl StreamingToolCallTracker { } /// Set the metadata extractor (the TS `extractMetadata` option). Called - /// once when a new tool call is detected. + /// once when a new tool call is detected; the value is kept with the call + /// and handed to the provider-metadata builder when the call finalizes. #[must_use] pub fn with_extract_metadata< - F: Fn(&StreamingToolCallDelta) -> Option + Send + Sync + 'static, + F: Fn(&StreamingToolCallDelta) -> Option + Send + Sync + 'static, >( mut self, f: F, @@ -254,9 +250,11 @@ impl StreamingToolCallTracker { /// Set the provider-metadata builder (the TS /// `buildToolCallProviderMetadata` option). If it returns `None`, the - /// `tool-call` part carries no `provider_metadata`. + /// `ToolCall` part carries no `provider_metadata`. #[must_use] - pub fn with_build_provider_metadata) -> Option + Send + Sync + 'static>( + pub fn with_build_provider_metadata< + F: Fn(Option<&Value>) -> Option + Send + Sync + 'static, + >( mut self, f: F, ) -> Self { @@ -276,7 +274,7 @@ impl StreamingToolCallTracker { pub fn process_delta( &mut self, delta: &StreamingToolCallDelta, - ) -> Result>, TrackerError> { + ) -> Result, TrackerError> { let wire_name = delta.function.as_ref().and_then(|f| f.name.as_deref()); let has_blank_name = wire_name.is_some_and(|n| n.trim().is_empty()); let wire_id = non_blank(delta.id.as_deref()); @@ -323,7 +321,7 @@ impl StreamingToolCallTracker { /// Finalize any unfinished tool calls and return the closing parts. Call /// once when the stream ends. - pub fn flush(&mut self) -> Vec> { + pub fn flush(&mut self) -> Vec { // Index order is only reliable when every call has an index; for // mixed streams keep insertion order. let mut order: Vec = (0..self.tool_calls.len()).collect(); @@ -475,7 +473,7 @@ impl StreamingToolCallTracker { wire_id: Option<&str>, index: Option, name: Option<&str>, - parts: &mut Vec>, + parts: &mut Vec, ) -> Result { match self.type_validation { TypeValidation::Required => { @@ -494,9 +492,13 @@ impl StreamingToolCallTracker { let name = name.ok_or(TrackerError::MissingFunctionName)?; let id = self.create_tool_call_id(wire_id)?; - parts.push(ToolCallStreamPart::ToolInputStart { + parts.push(StreamPart::ToolInputStart { id: id.clone(), tool_name: name.to_string(), + provider_executed: None, + dynamic: None, + title: None, + provider_metadata: None, }); let metadata = self @@ -526,9 +528,10 @@ impl StreamingToolCallTracker { } if !initial_arguments.is_empty() { - parts.push(ToolCallStreamPart::ToolInputDelta { + parts.push(StreamPart::ToolInputDelta { id, delta: initial_arguments, + provider_metadata: None, }); } @@ -541,7 +544,7 @@ impl StreamingToolCallTracker { &mut self, call: usize, arguments: Option<&str>, - parts: &mut Vec>, + parts: &mut Vec, ) { let tool_call = &mut self.tool_calls[call]; if tool_call.has_finished { @@ -550,9 +553,10 @@ impl StreamingToolCallTracker { if let Some(arguments) = arguments { tool_call.argument_state.append(arguments); tool_call.arguments.push_str(arguments); - parts.push(ToolCallStreamPart::ToolInputDelta { + parts.push(StreamPart::ToolInputDelta { id: tool_call.id.clone(), delta: arguments.to_string(), + provider_metadata: None, }); } } @@ -605,12 +609,13 @@ impl StreamingToolCallTracker { Err(TrackerError::IdExhausted) } - fn finish_tool_call(&mut self, call: usize, parts: &mut Vec>) { + fn finish_tool_call(&mut self, call: usize, parts: &mut Vec) { let tool_call = &mut self.tool_calls[call]; tool_call.has_finished = true; - parts.push(ToolCallStreamPart::ToolInputEnd { + parts.push(StreamPart::ToolInputEnd { id: tool_call.id.clone(), + provider_metadata: None, }); let provider_metadata = self @@ -618,10 +623,16 @@ impl StreamingToolCallTracker { .as_ref() .and_then(|build| build(tool_call.metadata.as_ref())); - parts.push(ToolCallStreamPart::ToolCall { + parts.push(StreamPart::ToolCall { tool_call_id: tool_call.id.clone(), tool_name: tool_call.function_name.clone(), - input: tool_call.arguments.clone(), + // Raw argument text; Core parses it after `stream_text`. + input: Value::String(tool_call.arguments.clone()), + provider_executed: None, + dynamic: None, + thought_signature: None, + invalid: None, + error: None, provider_metadata, }); } diff --git a/aimux-stream/tests/streaming_tool_call_argument_state_test.rs b/aimux-provider-utils/tests/streaming_tool_call_argument_state_test.rs similarity index 96% rename from aimux-stream/tests/streaming_tool_call_argument_state_test.rs rename to aimux-provider-utils/tests/streaming_tool_call_argument_state_test.rs index 4d7fba63..ea6d8707 100644 --- a/aimux-stream/tests/streaming_tool_call_argument_state_test.rs +++ b/aimux-provider-utils/tests/streaming_tool_call_argument_state_test.rs @@ -1,7 +1,7 @@ //! Port of `streaming-tool-call-argument-state.test.ts` from //! `@ai-sdk/provider-utils`. -use aimux_stream::{StreamingToolCallArgumentState, starts_with_structured_value}; +use aimux_provider_utils::{StreamingToolCallArgumentState, starts_with_structured_value}; #[test] fn starts_with_structured_value_true() { diff --git a/aimux-stream/tests/streaming_tool_call_tracker_test.rs b/aimux-provider-utils/tests/streaming_tool_call_tracker_test.rs similarity index 91% rename from aimux-stream/tests/streaming_tool_call_tracker_test.rs rename to aimux-provider-utils/tests/streaming_tool_call_tracker_test.rs index 7c55044c..238a0ecb 100644 --- a/aimux-stream/tests/streaming_tool_call_tracker_test.rs +++ b/aimux-provider-utils/tests/streaming_tool_call_tracker_test.rs @@ -1,35 +1,99 @@ //! Port of `streaming-tool-call-tracker.test.ts` from //! `@ai-sdk/provider-utils`, case for case. +//! +//! Tracker parts are projected to a local `Part` enum so the assertions stay +//! byte-for-byte comparable with the upstream expectations. use std::sync::Arc; use std::sync::atomic::{AtomicUsize, Ordering}; -use aimux_stream::{ - StreamingToolCallDelta, StreamingToolCallFunction, StreamingToolCallTracker, - ToolCallStreamPart, TrackerError, TypeValidation, +use aimux_core::stream_part::StreamPart; +use aimux_provider_utils::{ + StreamingToolCallDelta, StreamingToolCallFunction, StreamingToolCallTracker, TrackerError, + TypeValidation, }; use serde_json::{Value, json}; -type Part = ToolCallStreamPart; +/// The TS tracker's four events, projected out of [`StreamPart`] so the +/// assertions below read like the upstream suite. +#[derive(Debug, Clone, PartialEq)] +#[allow( + clippy::enum_variant_names, + reason = "named after the upstream stream parts" +)] +enum Part { + ToolInputStart { + id: String, + tool_name: String, + }, + ToolInputDelta { + id: String, + delta: String, + }, + ToolInputEnd { + id: String, + }, + ToolCall { + tool_call_id: String, + tool_name: String, + /// The raw argument text (`StreamPart::ToolCall.input` is a + /// `Value::String` from the tracker). + input: String, + provider_metadata: Option, + }, +} + +fn project(part: StreamPart) -> Part { + match part { + StreamPart::ToolInputStart { + id, + tool_name, + provider_executed: None, + dynamic: None, + title: None, + provider_metadata: None, + } => Part::ToolInputStart { id, tool_name }, + StreamPart::ToolInputDelta { + id, + delta, + provider_metadata: None, + } => Part::ToolInputDelta { id, delta }, + StreamPart::ToolInputEnd { + id, + provider_metadata: None, + } => Part::ToolInputEnd { id }, + StreamPart::ToolCall { + tool_call_id, + tool_name, + input: Value::String(input), + provider_executed: None, + dynamic: None, + thought_signature: None, + invalid: None, + error: None, + provider_metadata, + } => Part::ToolCall { + tool_call_id, + tool_name, + input, + provider_metadata, + }, + other => panic!("tracker emitted an unexpected part: {other:?}"), + } +} /// Tracker plus the parts it has emitted so far (the TS `createCollector`). -struct Harness { - tracker: StreamingToolCallTracker, - parts: Vec>, +struct Harness { + tracker: StreamingToolCallTracker, + parts: Vec, } -impl Harness<()> { +impl Harness { fn new() -> Self { Self::with(StreamingToolCallTracker::new()) } - fn with(tracker: StreamingToolCallTracker<()>) -> Self { - Self::with_meta(tracker) - } -} - -impl Harness { - fn with_meta(tracker: StreamingToolCallTracker) -> Self { + fn with(tracker: StreamingToolCallTracker) -> Self { Self { tracker, parts: Vec::new(), @@ -38,21 +102,19 @@ impl Harness { fn delta(&mut self, delta: StreamingToolCallDelta) -> Result<(), TrackerError> { let parts = self.tracker.process_delta(&delta)?; - self.parts.extend(parts); + self.parts.extend(parts.into_iter().map(project)); Ok(()) } fn flush(&mut self) { let parts = self.tracker.flush(); - self.parts.extend(parts); + self.parts.extend(parts.into_iter().map(project)); } fn clear(&mut self) { self.parts.clear(); } -} -impl Harness { /// `(id, name, input)` of every emitted `tool-call`, in order. fn tool_calls(&self) -> Vec<(String, String, String)> { self.parts @@ -992,7 +1054,7 @@ mod flush { mod metadata { use super::*; - fn google_tracker() -> StreamingToolCallTracker { + fn google_tracker() -> StreamingToolCallTracker { StreamingToolCallTracker::new() .with_extract_metadata(|delta| { delta.extra["extra_content"]["google"]["thought_signature"] @@ -1008,7 +1070,7 @@ mod metadata { #[test] fn extracts_and_includes_provider_metadata_in_tool_call_parts() { - let mut h = Harness::with_meta(google_tracker()); + let mut h = Harness::with(google_tracker()); h.delta( start(0, "call_1", "fn", "{}") @@ -1017,13 +1079,10 @@ mod metadata { .unwrap(); h.flush(); - let tool_call = h - .parts - .iter() - .find(|p| matches!(p, ToolCallStreamPart::ToolCall { .. })); + let tool_call = h.parts.iter().find(|p| matches!(p, Part::ToolCall { .. })); assert_eq!( tool_call, - Some(&ToolCallStreamPart::ToolCall { + Some(&Part::ToolCall { tool_call_id: "call_1".into(), tool_name: "fn".into(), input: "{}".into(), @@ -1034,7 +1093,7 @@ mod metadata { #[test] fn includes_provider_metadata_for_unfinished_tool_calls_finalized_in_flush() { - let mut h = Harness::with_meta( + let mut h = Harness::with( StreamingToolCallTracker::new() .with_extract_metadata(|_| Some(json!({ "custom": { "key": "value" } }))) .with_build_provider_metadata(|metadata| { @@ -1049,7 +1108,7 @@ mod metadata { assert_eq!( h.parts.last(), - Some(&ToolCallStreamPart::ToolCall { + Some(&Part::ToolCall { tool_call_id: "call_1".into(), tool_name: "fn".into(), input: "{\"incomplete".into(), @@ -1060,8 +1119,8 @@ mod metadata { #[test] fn omits_provider_metadata_when_the_builder_returns_none() { - let mut h = Harness::with_meta( - StreamingToolCallTracker::<()>::new() + let mut h = Harness::with( + StreamingToolCallTracker::new() .with_extract_metadata(|_| None) .with_build_provider_metadata(|_| None), ); @@ -1069,10 +1128,7 @@ mod metadata { h.delta(start(0, "call_1", "fn", "{}")).unwrap(); h.flush(); - let tool_call = h - .parts - .iter() - .find(|p| matches!(p, ToolCallStreamPart::ToolCall { .. })); + let tool_call = h.parts.iter().find(|p| matches!(p, Part::ToolCall { .. })); assert_eq!(tool_call, Some(&super::tool_call("call_1", "fn", "{}"))); } } diff --git a/aimux-providers/src/openai/model.rs b/aimux-providers/src/openai/model.rs index 6445d52c..107f984e 100644 --- a/aimux-providers/src/openai/model.rs +++ b/aimux-providers/src/openai/model.rs @@ -20,7 +20,9 @@ use aimux_core::result::{GenerateContent, GenerateResult, StreamResult}; use aimux_core::stream_part::StreamPart; use aimux_core::types::{FinishReason, FinishReasonUnified, ResponseMetadata, Usage}; -use aimux_provider_utils::HttpRequest; +use aimux_provider_utils::{ + HttpRequest, StreamingToolCallDelta, StreamingToolCallFunction, StreamingToolCallTracker, +}; use super::OpenAIConfig; use super::convert::{RequestBodyResult, build_request_body_with_warnings, parse_finish_reason}; @@ -176,15 +178,6 @@ fn convert_usage(usage: &UsageResponse, usage_raw: Option<&Value>) -> Usage { } } -// ── Tool-call accumulator (streaming) ──────────────────────────────────────── - -/// Accumulates a streamed tool call's id, name, and argument fragments. -struct ToolCallAccumulator { - id: String, - name: String, - arguments: String, -} - #[async_trait] impl LanguageModel for OpenAIModel { /// Provider identity for recording/routing. Uses `config.provider` (the @@ -491,9 +484,9 @@ pub async fn execute_stream( let mut response_metadata_emitted = false; let mut final_logprobs: Option = None; - // Tool-call accumulators keyed by OpenAI's `index` field. - let mut tool_calls: HashMap = HashMap::new(); - let mut tool_call_order: Vec = Vec::new(); + // Streamed tool calls, correlated by wire id, index and function name + // (the AI SDK's StreamingToolCallTracker) and finalized on flush. + let mut tool_calls = StreamingToolCallTracker::new(); // Process the first event (already peeked) then the rest. let mut event_iter = @@ -643,49 +636,30 @@ pub async fn execute_stream( reasoning_started = false; } for dtc in tool_call_deltas { - let idx = dtc.index; - let func = dtc.function.unwrap_or_default(); - - // New tool call: has id and/or name. - let is_new = !tool_calls.contains_key(&idx); - if is_new { - let id = dtc.id.unwrap_or_default(); - let name = func.name.unwrap_or_default(); - tool_calls.insert( - idx, - ToolCallAccumulator { - id: id.clone(), - name: name.clone(), - arguments: String::new(), - }, - ); - tool_call_order.push(idx); - yield Ok(StreamPart::ToolInputStart { - id, - tool_name: name, - provider_executed: None, - dynamic: None, - title: None, - provider_metadata: None, - }); - } - - // Argument delta. - // For new tool calls, skip the delta when - // arguments are empty (matches TS — the - // initial `""` is not emitted). For - // continuation chunks, always emit (even - // empty, matching TS). - if let Some(args) = func.arguments - && (!is_new || !args.is_empty()) - && let Some(acc) = tool_calls.get_mut(&idx) { - acc.arguments.push_str(&args); - yield Ok(StreamPart::ToolInputDelta { - id: acc.id.clone(), - delta: args, - provider_metadata: None, - }); + let delta = StreamingToolCallDelta { + index: dtc.index, + id: dtc.id, + r#type: None, + function: dtc.function.map(|f| StreamingToolCallFunction { + name: f.name, + arguments: f.arguments, + }), + extra: Value::Null, + }; + match tool_calls.process_delta(&delta) { + Ok(parts) => { + for part in parts { + yield Ok(part); } + } + // A malformed delta (new call without a + // function name) is invalid response + // data, as in the AI SDK; the stream ends. + Err(error) => { + yield Err(error.into()); + return; + } + } } } @@ -763,27 +737,10 @@ pub async fn execute_stream( }); } - // A parsable argument buffer can still be a prefix of a longer input. - // Match AI SDK's tracker by finalizing only when the stream flushes. - for &idx in &tool_call_order { - if let Some(acc) = tool_calls.get(&idx) { - yield Ok(StreamPart::ToolInputEnd { - id: acc.id.clone(), - provider_metadata: None, - }); - let input = Value::String(acc.arguments.clone()); - yield Ok(StreamPart::ToolCall { - tool_call_id: acc.id.clone(), - tool_name: acc.name.clone(), - input, - provider_executed: None, - dynamic: None, - thought_signature: None, - invalid: None, - error: None, - provider_metadata: None, - }); - } + // A parsable argument buffer can still be a prefix of a longer input: + // like the AI SDK's tracker, finalize only when the stream flushes. + for part in tool_calls.flush() { + yield Ok(part); } // Build provider metadata for the Finish part. diff --git a/aimux-providers/src/openai/types.rs b/aimux-providers/src/openai/types.rs index ea78af6e..1a8c8340 100644 --- a/aimux-providers/src/openai/types.rs +++ b/aimux-providers/src/openai/types.rs @@ -156,8 +156,10 @@ pub struct Delta { #[derive(Debug, Deserialize)] pub struct DeltaToolCall { + /// Optional: some OpenAI-compatible providers omit it (the AI SDK schema + /// is `index: z.number().nullish()`); the tracker falls back to id/name. #[serde(default)] - pub index: usize, + pub index: Option, #[serde(default)] pub id: Option, #[serde(default)] diff --git a/aimux-stream/Cargo.toml b/aimux-stream/Cargo.toml index 85b6bb20..4f167f01 100644 --- a/aimux-stream/Cargo.toml +++ b/aimux-stream/Cargo.toml @@ -4,7 +4,7 @@ version.workspace = true edition.workspace = true license.workspace = true description = "Streaming primitives for aimux" -keywords = ["llm", "streaming", "sse", "tool-calls"] +keywords = ["llm", "streaming", "sse"] categories = ["asynchronous", "parser-implementations"] documentation = "https://docs.rs/aimux-stream" @@ -12,11 +12,11 @@ documentation = "https://docs.rs/aimux-stream" futures = { workspace = true } pin-project-lite = { workspace = true } bytes = { workspace = true } -serde_json = { workspace = true } thiserror = { workspace = true } sse-stream = { version = "0.3.0", default-features = false, features = ["memchr"] } [dev-dependencies] +serde_json = { workspace = true } sha2 = "0.10" tokio = { workspace = true } diff --git a/aimux-stream/src/lib.rs b/aimux-stream/src/lib.rs index d9078d8d..3edc0c1a 100644 --- a/aimux-stream/src/lib.rs +++ b/aimux-stream/src/lib.rs @@ -1,21 +1,13 @@ //! # aimux-stream //! -//! Low-level streaming primitives for SSE (Server-Sent Events) parsing and -//! streamed tool-call tracking, -//! used by provider implementations to decode model API response streams. +//! Low-level SSE (Server-Sent Events) decoding, used by +//! `aimux-provider-utils` to decode model API response streams. The crate +//! plays the role `eventsource-parser` plays for the AI SDK: it knows nothing +//! about providers, models or stream parts. pub mod lines; pub mod sse; -pub mod streaming_tool_call_argument_state; -pub mod streaming_tool_call_tracker; // Re-export the most commonly used items. pub use lines::extract_lines; pub use sse::{SseError, SseEvent, SseStream}; -pub use streaming_tool_call_argument_state::{ - StreamingToolCallArgumentState, starts_with_structured_value, -}; -pub use streaming_tool_call_tracker::{ - StreamingToolCallDelta, StreamingToolCallFunction, StreamingToolCallTracker, - ToolCallStreamPart, TrackerError, TypeValidation, -}; diff --git a/docs/PROJECT-OVERVIEW.md b/docs/PROJECT-OVERVIEW.md index a53a672e..b79cb53a 100644 --- a/docs/PROJECT-OVERVIEW.md +++ b/docs/PROJECT-OVERVIEW.md @@ -165,8 +165,8 @@ aimux/ │ ├── native protocols # standalone model + convert, handles provider-specific differences │ ├── OpenAI compatible # registry-backed: provider-registry.json + provider(name, ...) entry (RFC-0017 phase 4) │ └── modalities/search # voice / image / video / search implementations -├── aimux-stream # SSE parsing, streamed tool-call tracking -├── aimux-provider-utils # One-exchange HTTP helpers, response handlers, API-key loading +├── aimux-stream # SSE decoding +├── aimux-provider-utils # One-exchange HTTP helpers, response handlers, API-key loading, streamed tool-call tracking ├── aimux-ffi # C ABI (FFI infrastructure, shared by all bindings) └── bindings/ # 6 language bindings ├── node/ # napi-rs v3 + typed TS wrapper