diff --git a/CHANGELOG.md b/CHANGELOG.md index 84c46819..b66e23f8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,6 +5,70 @@ All notable changes to aimux are documented in this file. The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). +## [Unreleased] + +### Breaking + +**Rust (aimux-stream)** + +- `StreamingToolCallTracker` rewritten to match the current AI SDK + (`@ai-sdk/provider-utils`) design; the previous version was a port of an + older, index-only tracker that nothing in the workspace used. Deltas are now + correlated by wire `id`, `index` and function name (ambiguous deltas are + dropped), ids are de-duplicated with bounded suffixes, blank function names + are ignored, and `flush` orders calls by index only when every call has one. + API changes: `process_delta` and `flush` now return the emitted + `ToolCallStreamPart`s instead of buffering them (`parts()` / `clear_parts()` + are gone); `with_max_index` and `TrackerError::{MissingId, IndexOutOfRange}` + are removed; `TrackerError::IdExhausted` is added; the builder closures must + be `Send + Sync`. +- Added `StreamingToolCallArgumentState` and `starts_with_structured_value`, + the structural JSON-prefix tracker the new correlation logic relies on. +- `SseStream` is now a thin adapter over the `sse-stream` crate (WHATWG + event-stream parsing), mirroring how the AI SDK's `parseJsonEventStream` + wraps `eventsource-parser`: `\n`, `\r` and `\r\n` line endings in any + mix, a leading UTF-8 BOM is stripped, a field line without `:` counts as an + empty value, an empty `event:` is `None`, an `id` containing U+0000 is + ignored, `retry` must be all ASCII digits, and a block is dispatched only + when it had a `data` line. There is no event size limit any more (as + upstream): `with_max_event_size` and `SseError::FrameTooLarge` are gone, + which fixes streams carrying multi-megabyte events (OpenAI Responses and + Gemini image generation). `SseError::Stream` carries the transport error + as its source instead of a `String`; `SseError::Utf8` holds a + `str::Utf8Error`; `SseError::Decode` is added. Decoder errors (invalid + UTF-8, transport failure) now end the stream after being reported. + `SseStream` requires the body error type to implement `std::error::Error + + Send + Sync + 'static`. +- Removed `NdjsonStream` / `NdjsonError`: nothing in the workspace used them + and the AI SDK has no counterpart. `tokio` is now a dev-dependency only and + the unused direct `serde` dependency is dropped. +- `StreamingToolCallTracker`, `StreamingToolCallArgumentState`, + `starts_with_structured_value` and `ToolCallStreamPart` are no longer in + `aimux-stream`, which is now SSE decoding only (the role + `eventsource-parser` plays for the AI SDK). + +**Rust (aimux-provider-utils)** + +- Gains `StreamingToolCallTracker` (plus `StreamingToolCallDelta`, + `StreamingToolCallFunction`, `TypeValidation`, `TrackerError`, + `StreamingToolCallArgumentState`), where `@ai-sdk/provider-utils` keeps it. + It emits `aimux_core::StreamPart` tool-input parts directly; the separate + `ToolCallStreamPart` event type is gone. `TrackerError` converts into + `AiMuxError::InvalidResponseData`. Metadata hooks are typed with + `serde_json::Value` / `ProviderMetadata` instead of a generic parameter. + +**Rust (aimux-providers)** + +- The OpenAI chat-completions stream (`openai/model.rs`, which serves every + registry-backed provider) correlates `tool_calls` deltas with the tracker + instead of by `index` alone: deltas are matched by wire id, index and + function name; a continuation without an id follows its call; indices + reused across parallel calls stay distinct; ambiguous deltas are dropped; + a call whose delta carries no id gets a generated `tool-call` / + `tool-call-N` id instead of an empty string; a new call without a function + name ends the stream with `InvalidResponseData` (previously it started a + call with an empty name). `DeltaToolCall.index` is now `Option`. + ## [0.5.0] - 2026-09-27 **Breaking release.** 13 PRs since 0.3.0: the cross-language error model diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index ce84ddbb..2b302fb8 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -9,8 +9,8 @@ to set up a development environment, run the tests, and submit changes. aimux/ ├── aimux-core/ # Core abstractions: LanguageModel / Provider / Message / StreamPart ├── aimux-providers/ # provider implementations + cassettes (counts: docs/api/providers.md) -├── aimux-stream/ # SSE / NDJSON stream parsing -├── aimux-provider-utils/ # HTTP utilities: retry, backoff, error parsing, API-key loading +├── aimux-stream/ # SSE decoding +├── aimux-provider-utils/ # HTTP utilities: retry, backoff, error parsing, API-key loading, streamed tool-call tracking ├── aimux-ffi/ # C ABI (opaque handle + JSON + push callback) for non-native bindings ├── bindings/ # Node, Python, Swift, Kotlin, Flutter, Go, C — share one Rust core ├── contract-tests/ # Shared JSON fixtures exercised across languages diff --git a/README.md b/README.md index e636bc1b..71fcfe0b 100644 --- a/README.md +++ b/README.md @@ -98,8 +98,8 @@ middleware, and telemetry per request). aimux/ ├── aimux-core # Core abstractions: LanguageModel / Provider / Message / StreamPart ├── aimux-providers # Provider implementations — registry-backed + typed (docs/api/providers.md) -├── aimux-stream # SSE / NDJSON stream parsing -├── aimux-provider-utils # One-exchange HTTP helpers, response handlers, API-key loading +├── aimux-stream # SSE decoding +├── aimux-provider-utils # One-exchange HTTP helpers, response handlers, API-key loading, streamed tool-call tracking ├── aimux-ffi # C ABI (opaque handles + JSON results + owned aimux_error_t *) for non-native bindings └── tools/ # aimux-cli (cache probe) · aimux-replay · aimux-web (console) ``` @@ -122,8 +122,8 @@ cargo add aimux-core aimux-providers |-------|-------------|-----------| | `aimux-core` | Core abstractions: `LanguageModel` / `Provider` / `Message` / `StreamPart` | [crates.io](https://crates.io/crates/aimux-core) | | `aimux-providers` | Provider implementations — [registry-backed + typed](docs/api/providers.md) | [crates.io](https://crates.io/crates/aimux-providers) | -| `aimux-stream` | SSE / NDJSON stream parsing | [crates.io](https://crates.io/crates/aimux-stream) | -| `aimux-provider-utils` | One-exchange HTTP helpers and typed response handlers | [crates.io](https://crates.io/crates/aimux-provider-utils) | +| `aimux-stream` | SSE decoding | [crates.io](https://crates.io/crates/aimux-stream) | +| `aimux-provider-utils` | One-exchange HTTP helpers, typed response handlers, streamed tool-call tracking | [crates.io](https://crates.io/crates/aimux-provider-utils) | | `aimux-ffi` | C ABI for non-native bindings | [crates.io](https://crates.io/crates/aimux-ffi) | **Node.js**: diff --git a/aimux-provider-utils/Cargo.toml b/aimux-provider-utils/Cargo.toml index 29a6049f..bf427a09 100644 --- a/aimux-provider-utils/Cargo.toml +++ b/aimux-provider-utils/Cargo.toml @@ -16,6 +16,7 @@ reqwest = { workspace = true } http = "1" serde = { workspace = true } serde_json = { workspace = true } +thiserror = { workspace = true } tokio = { workspace = true } tracing = { workspace = true } tracing-subscriber = { workspace = true } diff --git a/aimux-provider-utils/src/lib.rs b/aimux-provider-utils/src/lib.rs index 679f24be..fa33eb5d 100644 --- a/aimux-provider-utils/src/lib.rs +++ b/aimux-provider-utils/src/lib.rs @@ -3,7 +3,8 @@ //! Shared utilities for provider implementations. //! //! Provides one-exchange HTTP helpers, response handlers, API key loading, -//! header management, and URL utilities — the Rust equivalents of +//! header management, URL utilities and the streamed tool-call tracker for +//! the OpenAI chat-completions wire format — the Rust equivalents of //! `@ai-sdk/provider-utils`. Operation retry and timeout live in `aimux-core`. pub mod api_key; @@ -19,6 +20,8 @@ pub mod post_to_api; pub mod read_response_with_size_limit; pub mod response_handler; pub mod retry; +pub mod streaming_tool_call_argument_state; +pub mod streaming_tool_call_tracker; pub mod url; /// WebSocket client for realtime provider APIs (RFC-0028). Empty unless the /// `ws` feature is enabled. @@ -43,4 +46,11 @@ pub use response_handler::{ stream_error_api_call, }; pub use retry::RetryConfig; +pub use streaming_tool_call_argument_state::{ + StreamingToolCallArgumentState, starts_with_structured_value, +}; +pub use streaming_tool_call_tracker::{ + StreamingToolCallDelta, StreamingToolCallFunction, StreamingToolCallTracker, TrackerError, + TypeValidation, +}; pub use url::{validate_base_url, without_trailing_slash, without_trailing_slash_opt}; diff --git a/aimux-provider-utils/src/response_handler.rs b/aimux-provider-utils/src/response_handler.rs index db568ef8..2042669d 100644 --- a/aimux-provider-utils/src/response_handler.rs +++ b/aimux-provider-utils/src/response_handler.rs @@ -392,14 +392,14 @@ where match item { Ok(event) if event.data == "[DONE]" => continue, Ok(event) => yield serde_json::from_str::(&event.data).map_err(AiMuxError::from), - Err(aimux_stream::SseError::Stream(message)) => { + Err(aimux_stream::SseError::Stream(error)) => { // Preserve response transport failures as ApiCallError // items. Framing/parser failures remain JsonParse below. yield Err(AiMuxError::ApiCall(Box::new(ApiCallError { response_headers: Some(stream_headers.clone()), is_retryable: true, ..ApiCallError::new( - message, + error.to_string(), stream_url.clone(), stream_request_body_values.clone(), ) diff --git a/aimux-provider-utils/src/streaming_tool_call_argument_state.rs b/aimux-provider-utils/src/streaming_tool_call_argument_state.rs new file mode 100644 index 00000000..db156e66 --- /dev/null +++ b/aimux-provider-utils/src/streaming_tool_call_argument_state.rs @@ -0,0 +1,116 @@ +//! Incremental structure tracking for streamed tool-call arguments. +//! +//! Rust translation of `@ai-sdk/provider-utils`'s +//! `StreamingToolCallArgumentState` +//! (`packages/provider-utils/src/streaming-tool-call-argument-state.ts`). +//! +//! The state is intentionally *structural* rather than a JSON parse: a +//! currently parsable scalar can still be the prefix of a later value, but a +//! closed top-level `{…}` / `[…]` cannot be extended. + +#[derive(Debug, Clone, PartialEq, Eq)] +enum ArgumentStructure { + Undetermined, + Other, + Structured { + stack: Vec, + in_string: bool, + escaped: bool, + complete: bool, + }, +} + +/// Whether `value` (ignoring leading whitespace) starts with `{` or `[`. +#[must_use] +pub fn starts_with_structured_value(value: Option<&str>) -> bool { + value + .and_then(|v| v.trim_start().chars().next()) + .is_some_and(|c| c == '{' || c == '[') +} + +/// Incrementally tracks whether streamed tool-call arguments contain a +/// complete structured JSON value. +#[derive(Debug, Clone)] +pub struct StreamingToolCallArgumentState { + structure: ArgumentStructure, +} + +impl StreamingToolCallArgumentState { + /// Create a state seeded with `initial_value`. + #[must_use] + pub fn new(initial_value: &str) -> Self { + let mut state = Self { + structure: ArgumentStructure::Undetermined, + }; + state.append(initial_value); + state + } + + /// `true` once a top-level `{…}` / `[…]` value has been closed. + #[must_use] + pub fn has_complete_structured_value(&self) -> bool { + matches!( + self.structure, + ArgumentStructure::Structured { complete: true, .. } + ) + } + + /// Feed the next argument fragment. + pub fn append(&mut self, delta: &str) { + for character in delta.chars() { + match &mut self.structure { + ArgumentStructure::Undetermined => { + if character.is_whitespace() { + continue; + } + self.structure = if character == '{' || character == '[' { + ArgumentStructure::Structured { + stack: vec![character], + in_string: false, + escaped: false, + complete: false, + } + } else { + ArgumentStructure::Other + }; + } + ArgumentStructure::Other | ArgumentStructure::Structured { complete: true, .. } => { + } + ArgumentStructure::Structured { + stack, + in_string, + escaped, + complete, + } => { + if *in_string { + if *escaped { + *escaped = false; + } else if character == '\\' { + *escaped = true; + } else if character == '"' { + *in_string = false; + } + continue; + } + + match character { + '"' => *in_string = true, + '{' | '[' => stack.push(character), + '}' | ']' => { + let expected = if character == '}' { '{' } else { '[' }; + if stack.last() != Some(&expected) { + self.structure = ArgumentStructure::Other; + continue; + } + stack.pop(); + if stack.is_empty() { + *complete = true; + } + } + _ => {} + } + } + } + } + } +} diff --git a/aimux-provider-utils/src/streaming_tool_call_tracker.rs b/aimux-provider-utils/src/streaming_tool_call_tracker.rs new file mode 100644 index 00000000..ce9666dd --- /dev/null +++ b/aimux-provider-utils/src/streaming_tool_call_tracker.rs @@ -0,0 +1,649 @@ +//! Streaming tool call tracker. +//! +//! Rust translation of `@ai-sdk/provider-utils`'s `StreamingToolCallTracker` +//! (`packages/provider-utils/src/streaming-tool-call-tracker.ts`). +//! +//! Tracks streaming tool call state across the deltas of an OpenAI-compatible +//! chat completion stream: accumulates `arguments` fragments, emits +//! [`StreamPart::ToolInputStart`] / [`StreamPart::ToolInputDelta`] / +//! [`StreamPart::ToolInputEnd`] / [`StreamPart::ToolCall`] parts, and +//! finalizes unfinished calls on [`StreamingToolCallTracker::flush`]. +//! +//! Deltas are correlated to calls by wire `id`, `index` and function name +//! (see [`StreamingToolCallTracker`]'s resolution table), not by `index` +//! alone, so providers that reuse indices, repeat ids, drop ids on +//! continuations or send blank names are handled. +//! +//! Like the TS original, a call is *never* finalized before `flush`: a +//! parsable argument buffer can still be the prefix of a longer argument +//! string, so acting on it early would use truncated inputs (ai-sdk #13137). +//! +//! This is a tool for the OpenAI chat-completions wire format only. Protocols +//! whose streams carry explicit tool-call boundaries (Anthropic content +//! blocks, Google complete `functionCall` parts, Bedrock content blocks, +//! Cohere `tool-call-*` events, the Responses API's output items) do not need +//! it. + +use std::collections::{HashMap, HashSet}; + +use serde_json::Value; +use thiserror::Error; + +use aimux_core::error::AiMuxError; +use aimux_core::stream_part::StreamPart; +use aimux_core::types::ProviderMetadata; + +use crate::streaming_tool_call_argument_state::{ + StreamingToolCallArgumentState, starts_with_structured_value, +}; + +/// Fallback id used when the id generator returns a blank string. +const FALLBACK_TOOL_CALL_ID: &str = "tool-call"; + +/// The `function` sub-object of a streaming tool call delta. +#[derive(Debug, Clone, Default)] +pub struct StreamingToolCallFunction { + pub name: Option, + pub arguments: Option, +} + +/// A streaming tool call delta — the `tool_calls[i]` entry of an OpenAI-style +/// streaming chunk. +/// +/// `arguments: null` (TS) maps to `None`; `arguments: ''` maps to `Some("")`. +#[derive(Debug, Clone, Default)] +pub struct StreamingToolCallDelta { + pub index: Option, + pub id: Option, + /// The `type` field. Named `r#type` because `type` is a reserved word. + pub r#type: Option, + pub function: Option, + /// Provider-specific payload carried alongside the standard fields, read + /// by the `extract_metadata` hook (e.g. a Google thought signature). + pub extra: Value, +} + +impl StreamingToolCallDelta { + #[must_use] + pub fn new() -> Self { + Self::default() + } + + #[must_use] + pub fn index(mut self, index: usize) -> Self { + self.index = Some(index); + self + } + + #[must_use] + pub fn id(mut self, id: impl Into) -> Self { + self.id = Some(id.into()); + self + } + + /// Set the `type` field (named `tool_type` because `type` is reserved). + #[must_use] + pub fn tool_type(mut self, t: impl Into) -> Self { + self.r#type = Some(t.into()); + self + } + + #[must_use] + pub fn function_name(mut self, name: impl Into) -> Self { + self.function.get_or_insert_with(Default::default).name = Some(name.into()); + self + } + + /// Set the `function.arguments` fragment. Pass `""` for an explicit empty + /// fragment; omit the call for `None` (TS `arguments: null`). + #[must_use] + pub fn arguments(mut self, args: impl Into) -> Self { + self.function.get_or_insert_with(Default::default).arguments = Some(args.into()); + self + } + + #[must_use] + pub fn extra(mut self, extra: Value) -> Self { + self.extra = extra; + self + } +} + +/// How to validate the `type` field on a new tool call delta. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub enum TypeValidation { + /// No validation (default). + #[default] + None, + /// Error if `type` is present and not `"function"`. + IfPresent, + /// Error if `type` is not exactly `"function"`. + Required, +} + +/// Errors raised while processing a tool call delta. +/// +/// The TS tracker throws `InvalidResponseDataError`; these convert into +/// [`AiMuxError::InvalidResponseData`]. +#[derive(Debug, Error, PartialEq, Eq)] +pub enum TrackerError { + #[error("Expected 'function.name' to be a string.")] + MissingFunctionName, + #[error("Expected 'function' type.")] + InvalidType, + #[error("Failed to create a unique tool call ID.")] + IdExhausted, +} + +impl From for AiMuxError { + fn from(error: TrackerError) -> Self { + AiMuxError::InvalidResponseData(error.to_string()) + } +} + +struct TrackedToolCall { + id: String, + index: Option, + sequence: usize, + function_name: String, + arguments: String, + argument_state: StreamingToolCallArgumentState, + has_finished: bool, + metadata: Option, +} + +enum ToolCallResolution { + Existing(usize), + New, + Ambiguous, +} + +type GenerateIdFn = Box String + Send + Sync>; +type ExtractMetadataFn = Box Option + Send + Sync>; +type BuildMetadataFn = Box) -> Option + Send + Sync>; + +/// Tracks streaming tool call state across multiple deltas. +/// +/// [`process_delta`](Self::process_delta) and [`flush`](Self::flush) return +/// the [`StreamPart`]s to forward downstream (the TS tracker enqueues them on +/// a controller instead). +/// +/// # Correlation +/// +/// | ID evidence | index/name evidence | start evidence | resolution | +/// | --- | --- | --- | --- | +/// | known | matching | any | matching call, new call, or ambiguity | +/// | known | conflicting | named | new call | +/// | unseen | matching | structured start | new call | +/// | unseen | matching | continuation | matching call or ambiguity | +/// | absent | matching | any | matching call, new call, or ambiguity | +/// | absent | absent | named | new call | +/// | absent | absent | unnamed | sole unfinished call, new call, or ambiguity | +/// +/// An *ambiguous* delta is dropped. +pub struct StreamingToolCallTracker { + tool_calls: Vec, + tool_calls_by_id: HashMap>, + tool_calls_by_index: HashMap>, + used_tool_call_ids: HashSet, + next_generated_id_suffixes: HashMap, + generate_id: GenerateIdFn, + type_validation: TypeValidation, + extract_metadata: Option, + build_provider_metadata: Option, +} + +impl Default for StreamingToolCallTracker { + fn default() -> Self { + Self::new() + } +} + +impl StreamingToolCallTracker { + /// Create a tracker with no metadata handling and default settings. + /// + /// The default id generator returns a blank-free constant, so ids that + /// need generating become `tool-call`, `tool-call-1`, `tool-call-2`, … + #[must_use] + pub fn new() -> Self { + Self { + tool_calls: Vec::new(), + tool_calls_by_id: HashMap::new(), + tool_calls_by_index: HashMap::new(), + used_tool_call_ids: HashSet::new(), + next_generated_id_suffixes: HashMap::new(), + generate_id: Box::new(|| FALLBACK_TOOL_CALL_ID.to_string()), + type_validation: TypeValidation::None, + extract_metadata: None, + build_provider_metadata: None, + } + } + + /// Set the id generator (the TS `generateId` option). Blank or repeated + /// outputs are turned into usable unique ids. + #[must_use] + pub fn with_generate_id String + Send + Sync + 'static>(mut self, f: F) -> Self { + self.generate_id = Box::new(f); + self + } + + /// Set the `type` validation mode (the TS `typeValidation` option). + #[must_use] + pub fn with_type_validation(mut self, v: TypeValidation) -> Self { + self.type_validation = v; + self + } + + /// Set the metadata extractor (the TS `extractMetadata` option). Called + /// once when a new tool call is detected; the value is kept with the call + /// and handed to the provider-metadata builder when the call finalizes. + #[must_use] + pub fn with_extract_metadata< + F: Fn(&StreamingToolCallDelta) -> Option + Send + Sync + 'static, + >( + mut self, + f: F, + ) -> Self { + self.extract_metadata = Some(Box::new(f)); + self + } + + /// Set the provider-metadata builder (the TS + /// `buildToolCallProviderMetadata` option). If it returns `None`, the + /// `ToolCall` part carries no `provider_metadata`. + #[must_use] + pub fn with_build_provider_metadata< + F: Fn(Option<&Value>) -> Option + Send + Sync + 'static, + >( + mut self, + f: F, + ) -> Self { + self.build_provider_metadata = Some(Box::new(f)); + self + } + + /// Process a tool call delta from a streaming chunk and return the parts + /// it produces (possibly none). + /// + /// # Errors + /// + /// Returns [`TrackerError::InvalidType`] when `type` validation fails, + /// [`TrackerError::MissingFunctionName`] when a new call has no + /// (non-null) function name, and [`TrackerError::IdExhausted`] if no + /// unique id can be produced. + pub fn process_delta( + &mut self, + delta: &StreamingToolCallDelta, + ) -> Result, TrackerError> { + let wire_name = delta.function.as_ref().and_then(|f| f.name.as_deref()); + let has_blank_name = wire_name.is_some_and(|n| n.trim().is_empty()); + let wire_id = non_blank(delta.id.as_deref()); + let name = non_blank(wire_name); + let index = delta.index; + let arguments = delta.function.as_ref().and_then(|f| f.arguments.as_deref()); + + let resolution = self.resolve_tool_call( + wire_id, + index, + name, + name.is_some() && starts_with_structured_value(arguments), + ); + + let mut parts = Vec::new(); + let call = match resolution { + ToolCallResolution::Ambiguous => return Ok(parts), + ToolCallResolution::New => { + // Blank names cannot start a usable call, but some providers + // repeat a blank name on continuations. Those were correlated + // above; only an unmatched blank-name delta is ignored. + if has_blank_name { + return Ok(parts); + } + self.process_new_tool_call(delta, wire_id, index, name, &mut parts)? + } + ToolCallResolution::Existing(call) => { + if let Some(wire_id) = wire_id { + self.associate_wire_id(call, wire_id); + } + self.process_existing_tool_call(call, arguments, &mut parts); + call + } + }; + + if let Some(index) = index { + self.tool_calls_by_index + .entry(index) + .or_default() + .insert(call); + } + Ok(parts) + } + + /// Finalize any unfinished tool calls and return the closing parts. Call + /// once when the stream ends. + pub fn flush(&mut self) -> Vec { + // Index order is only reliable when every call has an index; for + // mixed streams keep insertion order. + let mut order: Vec = (0..self.tool_calls.len()).collect(); + if self.tool_calls.iter().all(|c| c.index.is_some()) { + order.sort_by_key(|&i| (self.tool_calls[i].index, self.tool_calls[i].sequence)); + } + + let mut parts = Vec::new(); + for call in order { + if !self.tool_calls[call].has_finished { + self.finish_tool_call(call, &mut parts); + } + } + parts + } + + fn resolve_tool_call( + &self, + wire_id: Option<&str>, + index: Option, + name: Option<&str>, + has_explicit_call_start: bool, + ) -> ToolCallResolution { + let indexed = index.and_then(|i| self.tool_calls_by_index.get(&i)); + let matching_indexed = self.filter_by_name(indexed, name); + + if let Some(wire_id) = wire_id { + if let Some(with_id) = self.tool_calls_by_id.get(wire_id) { + if index.is_some() { + let matching: Vec = matching_indexed + .iter() + .copied() + .filter(|c| with_id.contains(c)) + .collect(); + let resolved = self.resolve_matching(matching, has_explicit_call_start); + if !matches!(resolved, ToolCallResolution::New) { + return resolved; + } + + // A named delta with a distinct index starts a new call + // even when its wire id and name repeat: providers may + // reuse ids across parallel calls. + if name.is_some() { + return ToolCallResolution::New; + } + + // Conflicting labels on a continuation cannot be + // resolved safely. + if indexed.is_some() { + return ToolCallResolution::Ambiguous; + } + + return self.resolve_matching(sorted(with_id), false); + } + + if let Some(name) = name { + let matching: Vec = sorted(with_id) + .into_iter() + .filter(|&c| self.tool_calls[c].function_name == name) + .collect(); + return self.resolve_matching(matching, has_explicit_call_start); + } + + return self.resolve_matching(sorted(with_id), false); + } + + if !matching_indexed.is_empty() { + // A previously unseen id plus a named structured argument + // start is stronger evidence of a distinct call than a reused + // index/name; ids may still change on plain continuations. + return if has_explicit_call_start { + ToolCallResolution::New + } else { + self.resolve_matching(matching_indexed, false) + }; + } + + return ToolCallResolution::New; + } + + if indexed.is_some() { + // Repeated names are valid on continuations; a different name at + // the same index means a new call from a provider that reuses + // indices across parallel calls. + return self.resolve_matching(matching_indexed, has_explicit_call_start); + } + + if name.is_some() { + return ToolCallResolution::New; + } + + let unfinished: Vec = (0..self.tool_calls.len()) + .filter(|&c| !self.tool_calls[c].has_finished) + .collect(); + match unfinished.len() { + 0 => ToolCallResolution::New, + 1 => ToolCallResolution::Existing(unfinished[0]), + _ => ToolCallResolution::Ambiguous, + } + } + + fn filter_by_name(&self, calls: Option<&HashSet>, name: Option<&str>) -> Vec { + calls.map_or_else(Vec::new, |calls| { + sorted(calls) + .into_iter() + .filter(|&c| name.is_none_or(|n| self.tool_calls[c].function_name == n)) + .collect() + }) + } + + fn resolve_matching( + &self, + calls: Vec, + has_explicit_call_start: bool, + ) -> ToolCallResolution { + if calls.is_empty() { + return ToolCallResolution::New; + } + + if !has_explicit_call_start { + return if calls.len() == 1 { + ToolCallResolution::Existing(calls[0]) + } else { + ToolCallResolution::Ambiguous + }; + } + + // A repeated name can occur on continuations. A fresh structured + // argument prefix signals another call only once the matching call + // has completed its own structured payload. + let continuable: Vec = calls + .into_iter() + .filter(|&c| { + !self.tool_calls[c] + .argument_state + .has_complete_structured_value() + }) + .collect(); + match continuable.len() { + 0 => ToolCallResolution::New, + 1 => ToolCallResolution::Existing(continuable[0]), + _ => ToolCallResolution::Ambiguous, + } + } + + fn process_new_tool_call( + &mut self, + delta: &StreamingToolCallDelta, + wire_id: Option<&str>, + index: Option, + name: Option<&str>, + parts: &mut Vec, + ) -> Result { + match self.type_validation { + TypeValidation::Required => { + if delta.r#type.as_deref() != Some("function") { + return Err(TrackerError::InvalidType); + } + } + TypeValidation::IfPresent => { + if delta.r#type.as_deref().is_some_and(|t| t != "function") { + return Err(TrackerError::InvalidType); + } + } + TypeValidation::None => {} + } + + let name = name.ok_or(TrackerError::MissingFunctionName)?; + let id = self.create_tool_call_id(wire_id)?; + + parts.push(StreamPart::ToolInputStart { + id: id.clone(), + tool_name: name.to_string(), + provider_executed: None, + dynamic: None, + title: None, + provider_metadata: None, + }); + + let metadata = self + .extract_metadata + .as_ref() + .and_then(|extract| extract(delta)); + + let initial_arguments = delta + .function + .as_ref() + .and_then(|f| f.arguments.clone()) + .unwrap_or_default(); + + let call = self.tool_calls.len(); + self.tool_calls.push(TrackedToolCall { + id: id.clone(), + index, + sequence: call, + function_name: name.to_string(), + argument_state: StreamingToolCallArgumentState::new(&initial_arguments), + arguments: initial_arguments.clone(), + has_finished: false, + metadata, + }); + if let Some(wire_id) = wire_id { + self.associate_wire_id(call, wire_id); + } + + if !initial_arguments.is_empty() { + parts.push(StreamPart::ToolInputDelta { + id, + delta: initial_arguments, + provider_metadata: None, + }); + } + + // Tool calls must not finalize before the stream ends (ai-sdk + // #13137); finalization happens in `flush`. + Ok(call) + } + + fn process_existing_tool_call( + &mut self, + call: usize, + arguments: Option<&str>, + parts: &mut Vec, + ) { + let tool_call = &mut self.tool_calls[call]; + if tool_call.has_finished { + return; + } + if let Some(arguments) = arguments { + tool_call.argument_state.append(arguments); + tool_call.arguments.push_str(arguments); + parts.push(StreamPart::ToolInputDelta { + id: tool_call.id.clone(), + delta: arguments.to_string(), + provider_metadata: None, + }); + } + } + + fn associate_wire_id(&mut self, call: usize, wire_id: &str) { + self.tool_calls_by_id + .entry(wire_id.to_string()) + .or_default() + .insert(call); + } + + fn create_tool_call_id(&mut self, wire_id: Option<&str>) -> Result { + if let Some(wire_id) = wire_id + && !self.used_tool_call_ids.contains(wire_id) + { + self.used_tool_call_ids.insert(wire_id.to_string()); + return Ok(wire_id.to_string()); + } + + let generated = non_blank(Some(&(self.generate_id)())) + .unwrap_or(FALLBACK_TOOL_CALL_ID) + .to_string(); + + if !self.used_tool_call_ids.contains(&generated) { + self.used_tool_call_ids.insert(generated.clone()); + return Ok(generated); + } + + // Resume after the last suffix checked for this generated value so + // deterministic generators stay bounded without rescanning occupied + // suffixes. + let initial_suffix = self + .next_generated_id_suffixes + .get(&generated) + .copied() + .unwrap_or(1); + let maximum_suffix = initial_suffix + self.used_tool_call_ids.len(); + for suffix in initial_suffix..=maximum_suffix { + let suffixed = format!("{generated}-{suffix}"); + if !self.used_tool_call_ids.contains(&suffixed) { + self.used_tool_call_ids.insert(suffixed.clone()); + self.next_generated_id_suffixes + .insert(generated, suffix + 1); + return Ok(suffixed); + } + } + + // Unreachable by the pigeonhole principle; guards the invariant + // instead of looping without bound. + Err(TrackerError::IdExhausted) + } + + fn finish_tool_call(&mut self, call: usize, parts: &mut Vec) { + let tool_call = &mut self.tool_calls[call]; + tool_call.has_finished = true; + + parts.push(StreamPart::ToolInputEnd { + id: tool_call.id.clone(), + provider_metadata: None, + }); + + let provider_metadata = self + .build_provider_metadata + .as_ref() + .and_then(|build| build(tool_call.metadata.as_ref())); + + parts.push(StreamPart::ToolCall { + tool_call_id: tool_call.id.clone(), + tool_name: tool_call.function_name.clone(), + // Raw argument text; Core parses it after `stream_text`. + input: Value::String(tool_call.arguments.clone()), + provider_executed: None, + dynamic: None, + thought_signature: None, + invalid: None, + error: None, + provider_metadata, + }); + } +} + +fn non_blank(value: Option<&str>) -> Option<&str> { + value.filter(|v| !v.trim().is_empty()) +} + +fn sorted(calls: &HashSet) -> Vec { + let mut calls: Vec = calls.iter().copied().collect(); + calls.sort_unstable(); + calls +} diff --git a/aimux-provider-utils/tests/streaming_tool_call_argument_state_test.rs b/aimux-provider-utils/tests/streaming_tool_call_argument_state_test.rs new file mode 100644 index 00000000..ea6d8707 --- /dev/null +++ b/aimux-provider-utils/tests/streaming_tool_call_argument_state_test.rs @@ -0,0 +1,74 @@ +//! Port of `streaming-tool-call-argument-state.test.ts` from +//! `@ai-sdk/provider-utils`. + +use aimux_provider_utils::{StreamingToolCallArgumentState, starts_with_structured_value}; + +#[test] +fn starts_with_structured_value_true() { + for value in ["{}", " {", "[]", "\n["] { + assert!(starts_with_structured_value(Some(value)), "{value:?}"); + } +} + +#[test] +fn starts_with_structured_value_false() { + // TS covers `undefined` and `null` separately; Rust has one `None`. + assert!(!starts_with_structured_value(None)); + for value in ["", " ", "1", "\"value\""] { + assert!(!starts_with_structured_value(Some(value)), "{value:?}"); + } +} + +#[test] +fn tracks_a_structured_value_across_deltas() { + let mut state = StreamingToolCallArgumentState::new(" {\"value\":"); + assert!(!state.has_complete_structured_value()); + + state.append("1}"); + assert!(state.has_complete_structured_value()); +} + +#[test] +fn tracks_nested_objects_and_arrays() { + let state = StreamingToolCallArgumentState::new("[{\"value\":{\"items\":[1,2]}}]"); + assert!(state.has_complete_structured_value()); +} + +#[test] +fn ignores_structural_characters_inside_strings() { + let state = StreamingToolCallArgumentState::new("{\"value\":\"braces: } ] { [\"}"); + assert!(state.has_complete_structured_value()); +} + +#[test] +fn handles_escaped_quotes_across_deltas() { + let mut state = StreamingToolCallArgumentState::new("{\"value\":\"escaped quote: \\\""); + assert!(!state.has_complete_structured_value()); + + state.append(" still in string\"}"); + assert!(state.has_complete_structured_value()); +} + +#[test] +fn can_begin_after_an_empty_or_whitespace_only_delta() { + let mut state = StreamingToolCallArgumentState::new(" "); + + state.append("["); + assert!(!state.has_complete_structured_value()); + + state.append("]"); + assert!(state.has_complete_structured_value()); +} + +#[test] +fn does_not_treat_scalar_arguments_as_a_complete_structured_value() { + let state = StreamingToolCallArgumentState::new("12"); + assert!(!state.has_complete_structured_value()); +} + +#[test] +fn does_not_recover_mismatched_structures_as_complete() { + let mut state = StreamingToolCallArgumentState::new("{\"value\":]"); + state.append("}"); + assert!(!state.has_complete_structured_value()); +} diff --git a/aimux-provider-utils/tests/streaming_tool_call_tracker_test.rs b/aimux-provider-utils/tests/streaming_tool_call_tracker_test.rs new file mode 100644 index 00000000..238a0ecb --- /dev/null +++ b/aimux-provider-utils/tests/streaming_tool_call_tracker_test.rs @@ -0,0 +1,1154 @@ +//! Port of `streaming-tool-call-tracker.test.ts` from +//! `@ai-sdk/provider-utils`, case for case. +//! +//! Tracker parts are projected to a local `Part` enum so the assertions stay +//! byte-for-byte comparable with the upstream expectations. + +use std::sync::Arc; +use std::sync::atomic::{AtomicUsize, Ordering}; + +use aimux_core::stream_part::StreamPart; +use aimux_provider_utils::{ + StreamingToolCallDelta, StreamingToolCallFunction, StreamingToolCallTracker, TrackerError, + TypeValidation, +}; +use serde_json::{Value, json}; + +/// The TS tracker's four events, projected out of [`StreamPart`] so the +/// assertions below read like the upstream suite. +#[derive(Debug, Clone, PartialEq)] +#[allow( + clippy::enum_variant_names, + reason = "named after the upstream stream parts" +)] +enum Part { + ToolInputStart { + id: String, + tool_name: String, + }, + ToolInputDelta { + id: String, + delta: String, + }, + ToolInputEnd { + id: String, + }, + ToolCall { + tool_call_id: String, + tool_name: String, + /// The raw argument text (`StreamPart::ToolCall.input` is a + /// `Value::String` from the tracker). + input: String, + provider_metadata: Option, + }, +} + +fn project(part: StreamPart) -> Part { + match part { + StreamPart::ToolInputStart { + id, + tool_name, + provider_executed: None, + dynamic: None, + title: None, + provider_metadata: None, + } => Part::ToolInputStart { id, tool_name }, + StreamPart::ToolInputDelta { + id, + delta, + provider_metadata: None, + } => Part::ToolInputDelta { id, delta }, + StreamPart::ToolInputEnd { + id, + provider_metadata: None, + } => Part::ToolInputEnd { id }, + StreamPart::ToolCall { + tool_call_id, + tool_name, + input: Value::String(input), + provider_executed: None, + dynamic: None, + thought_signature: None, + invalid: None, + error: None, + provider_metadata, + } => Part::ToolCall { + tool_call_id, + tool_name, + input, + provider_metadata, + }, + other => panic!("tracker emitted an unexpected part: {other:?}"), + } +} + +/// Tracker plus the parts it has emitted so far (the TS `createCollector`). +struct Harness { + tracker: StreamingToolCallTracker, + parts: Vec, +} + +impl Harness { + fn new() -> Self { + Self::with(StreamingToolCallTracker::new()) + } + + fn with(tracker: StreamingToolCallTracker) -> Self { + Self { + tracker, + parts: Vec::new(), + } + } + + fn delta(&mut self, delta: StreamingToolCallDelta) -> Result<(), TrackerError> { + let parts = self.tracker.process_delta(&delta)?; + self.parts.extend(parts.into_iter().map(project)); + Ok(()) + } + + fn flush(&mut self) { + let parts = self.tracker.flush(); + self.parts.extend(parts.into_iter().map(project)); + } + + fn clear(&mut self) { + self.parts.clear(); + } + + /// `(id, name, input)` of every emitted `tool-call`, in order. + fn tool_calls(&self) -> Vec<(String, String, String)> { + self.parts + .iter() + .filter_map(|part| match part { + Part::ToolCall { + tool_call_id, + tool_name, + input, + .. + } => Some((tool_call_id.clone(), tool_name.clone(), input.clone())), + _ => None, + }) + .collect() + } +} + +/// A delta with an explicit `function` object (as every TS test passes one). +fn delta( + index: Option, + id: Option<&str>, + ty: Option<&str>, + name: Option<&str>, + arguments: Option<&str>, +) -> StreamingToolCallDelta { + StreamingToolCallDelta { + index, + id: id.map(str::to_string), + r#type: ty.map(str::to_string), + function: Some(StreamingToolCallFunction { + name: name.map(str::to_string), + arguments: arguments.map(str::to_string), + }), + extra: Value::Null, + } +} + +/// Full call start: index, id, `type: 'function'`, name and arguments. +fn start(index: usize, id: &str, name: &str, arguments: &str) -> StreamingToolCallDelta { + delta( + Some(index), + Some(id), + Some("function"), + Some(name), + Some(arguments), + ) +} + +/// Continuation carrying only arguments (plus an optional index). +fn cont(index: Option, arguments: &str) -> StreamingToolCallDelta { + delta(index, None, None, None, Some(arguments)) +} + +fn tc(id: &str, name: &str, input: &str) -> (String, String, String) { + (id.into(), name.into(), input.into()) +} + +fn input_start(id: &str, tool_name: &str) -> Part { + Part::ToolInputStart { + id: id.into(), + tool_name: tool_name.into(), + } +} + +fn input_delta(id: &str, delta: &str) -> Part { + Part::ToolInputDelta { + id: id.into(), + delta: delta.into(), + } +} + +fn input_end(id: &str) -> Part { + Part::ToolInputEnd { id: id.into() } +} + +fn tool_call(id: &str, name: &str, input: &str) -> Part { + Part::ToolCall { + tool_call_id: id.into(), + tool_name: name.into(), + input: input.into(), + provider_metadata: None, + } +} + +/// Deterministic id generator: `prefix-1`, `prefix-2`, … and a call counter. +fn counting_generator(ids: &'static [&'static str]) -> (impl Fn() -> String, Arc) { + let calls = Arc::new(AtomicUsize::new(0)); + let counter = Arc::clone(&calls); + ( + move || { + let n = counter.fetch_add(1, Ordering::SeqCst); + ids[n.min(ids.len() - 1)].to_string() + }, + calls, + ) +} + +mod process_delta { + use super::*; + + #[test] + fn single_tool_call_accumulated_across_multiple_deltas() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "get_weather", "{\"ci")).unwrap(); + assert_eq!( + h.parts, + vec![ + input_start("call_1", "get_weather"), + input_delta("call_1", "{\"ci"), + ] + ); + h.clear(); + + h.delta(cont(Some(0), "ty\": \"San")).unwrap(); + assert_eq!(h.parts, vec![input_delta("call_1", "ty\": \"San")]); + h.clear(); + + // Completing the JSON must not finalize before flush: a parsable + // buffer can still be the prefix of longer arguments. + h.delta(cont(Some(0), " Francisco\"}")).unwrap(); + assert_eq!(h.parts, vec![input_delta("call_1", " Francisco\"}")]); + h.clear(); + + h.flush(); + assert_eq!( + h.parts, + vec![ + input_end("call_1"), + tool_call("call_1", "get_weather", "{\"city\": \"San Francisco\"}"), + ] + ); + } + + #[test] + fn full_tool_call_in_a_single_chunk() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "get_weather", "{\"city\": \"London\"}")) + .unwrap(); + assert_eq!( + h.parts, + vec![ + input_start("call_1", "get_weather"), + input_delta("call_1", "{\"city\": \"London\"}"), + ] + ); + h.clear(); + + h.flush(); + assert_eq!( + h.parts, + vec![ + input_end("call_1"), + tool_call("call_1", "get_weather", "{\"city\": \"London\"}"), + ] + ); + } + + #[test] + fn does_not_finalize_when_argument_prefix_is_parsable_json() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "search", "{\"query\": \"test\"}")) + .unwrap(); + assert_eq!( + h.parts, + vec![ + input_start("call_1", "search"), + input_delta("call_1", "{\"query\": \"test\"}"), + ] + ); + + h.delta(cont(Some(0), ", \"limit\": 10}")).unwrap(); + h.flush(); + + assert_eq!( + h.parts.last(), + Some(&tool_call( + "call_1", + "search", + "{\"query\": \"test\"}, \"limit\": 10}" + )) + ); + assert_eq!(h.tool_calls().len(), 1); + } + + #[test] + fn multiple_concurrent_tool_calls() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "get_weather", "")).unwrap(); + h.delta(start(1, "call_2", "get_time", "")).unwrap(); + + assert_eq!( + h.parts, + vec![ + input_start("call_1", "get_weather"), + input_start("call_2", "get_time"), + ] + ); + } + + #[test] + fn non_zero_and_non_contiguous_indexes() { + let mut h = Harness::new(); + + h.delta(start(1, "call_1", "fn1", "{\"value\":1}")).unwrap(); + h.delta(start(3, "call_2", "fn2", "{\"value\":2}")).unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![ + tc("call_1", "fn1", "{\"value\":1}"), + tc("call_2", "fn2", "{\"value\":2}"), + ] + ); + } + + #[test] + fn keeps_distinct_tool_calls_that_reuse_an_index() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "fn", "{\"value\":1}")).unwrap(); + h.delta(start(0, "call_2", "fn", "{\"value\":2}")).unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![ + tc("call_1", "fn", "{\"value\":1}"), + tc("call_2", "fn", "{\"value\":2}"), + ] + ); + } + + #[test] + fn continues_latest_call_when_index_is_omitted_after_starting_at() { + for index in [None, Some(7)] { + let mut h = Harness::new(); + + h.delta(delta( + index, + Some("call_1"), + Some("function"), + Some("fn"), + Some("{\"val"), + )) + .unwrap(); + h.delta(cont(None, "ue\":1}")).unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![tc("call_1", "fn", "{\"value\":1}")], + "index {index:?}" + ); + } + } + + #[test] + fn uses_the_index_when_continuation_ids_are_empty() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "fn", "{\"val")).unwrap(); + h.delta(delta( + Some(0), + Some(""), + Some("function"), + None, + Some("ue\":1}"), + )) + .unwrap(); + h.flush(); + + assert_eq!(h.tool_calls(), vec![tc("call_1", "fn", "{\"value\":1}")]); + } + + #[test] + fn skips_deltas_for_already_finished_tool_calls() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "fn", "{}")).unwrap(); + h.flush(); + h.clear(); + + h.delta(cont(Some(0), "extra")).unwrap(); + assert_eq!(h.parts, vec![]); + } + + #[test] + fn skips_delta_emission_when_arguments_are_null() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "fn", "")).unwrap(); + h.clear(); + + h.delta(delta(Some(0), None, None, None, None)).unwrap(); + assert_eq!(h.parts, vec![]); + } + + #[test] + fn uses_index_fallback_when_index_is_not_provided() { + let mut h = Harness::new(); + + h.delta(delta( + None, + Some("call_1"), + Some("function"), + Some("fn1"), + Some("{}"), + )) + .unwrap(); + h.delta(delta( + None, + Some("call_2"), + Some("function"), + Some("fn2"), + Some("{}"), + )) + .unwrap(); + + let starts: Vec<&Part> = h + .parts + .iter() + .filter(|p| matches!(p, Part::ToolInputStart { .. })) + .collect(); + assert_eq!( + starts, + vec![&input_start("call_1", "fn1"), &input_start("call_2", "fn2")] + ); + } + + #[test] + fn generates_an_id_when_id_is_missing() { + let mut h = Harness::with( + StreamingToolCallTracker::new().with_generate_id(|| "generated-id".to_string()), + ); + + h.delta(delta( + Some(0), + None, + Some("function"), + Some("fn"), + Some("{}"), + )) + .unwrap(); + h.flush(); + + assert_eq!(h.tool_calls(), vec![tc("generated-id", "fn", "{}")]); + } + + #[test] + fn errors_when_function_name_is_missing() { + // TS covers `name: undefined` and `name: null`; both are `None`. + let mut h = Harness::new(); + + let result = h.delta(delta(Some(0), Some("call_1"), Some("function"), None, None)); + + assert_eq!(result, Err(TrackerError::MissingFunctionName)); + assert_eq!( + TrackerError::MissingFunctionName.to_string(), + "Expected 'function.name' to be a string." + ); + } + + #[test] + fn ignores_a_blank_function_name_without_preventing_prior_calls_from_finalizing() { + for name in ["", " "] { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "valid_tool", "{\"value\":1}")) + .unwrap(); + h.delta(start(1, "call_2", name, "{\"value\":2}")).unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![tc("call_1", "valid_tool", "{\"value\":1}")], + "name {name:?}" + ); + } + } + + #[test] + fn retains_continuation_arguments_for_a_blank_name() { + // (blank name, continuation id, continuation index) + let cases = [ + ("", Some("call_1"), None), // blank name with a matching id + (" ", None, Some(0)), // whitespace name, matching index + ]; + for (name, id, index) in cases { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "read_file", "{\"pa")).unwrap(); + h.delta(delta(index, id, None, Some(name), Some("th\":\"a\"}"))) + .unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![tc("call_1", "read_file", "{\"path\":\"a\"}")], + "name {name:?}" + ); + } + } + + #[test] + fn keeps_id_less_calls_distinct_when_an_index_is_reused_and_type_is_omitted() { + let (generate, _) = counting_generator(&["generated-1", "generated-2", "generated-3"]); + let mut h = Harness::with(StreamingToolCallTracker::new().with_generate_id(generate)); + + h.delta(delta( + Some(0), + None, + None, + Some("read_file"), + Some("{\"path\":\"p0\"}"), + )) + .unwrap(); + h.delta(delta( + Some(0), + None, + None, + Some("write_file"), + Some("{\"path\":\"p1\"}"), + )) + .unwrap(); + h.delta(delta( + Some(0), + None, + None, + Some("read_file"), + Some("{\"path\":\"p2\"}"), + )) + .unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![ + tc("generated-1", "read_file", "{\"path\":\"p0\"}"), + tc("generated-2", "write_file", "{\"path\":\"p1\"}"), + tc("generated-3", "read_file", "{\"path\":\"p2\"}"), + ] + ); + } + + #[test] + fn keeps_complete_same_name_calls_distinct_with_reused_index() { + for id in [None, Some("dup")] { + let mut h = Harness::with( + StreamingToolCallTracker::new().with_generate_id(|| "generated-id".to_string()), + ); + + for value in ["{\"value\":1}", "{\"value\":2}"] { + h.delta(delta( + Some(0), + id, + Some("function"), + Some("same_tool"), + Some(value), + )) + .unwrap(); + } + h.flush(); + + let calls = h.tool_calls(); + let inputs: Vec<&str> = calls.iter().map(|c| c.2.as_str()).collect(); + assert_eq!(inputs, vec!["{\"value\":1}", "{\"value\":2}"], "id {id:?}"); + let ids: std::collections::HashSet<&str> = calls.iter().map(|c| c.0.as_str()).collect(); + assert_eq!(ids.len(), 2, "id {id:?}"); + } + } + + #[test] + fn keeps_a_partial_same_name_call_distinct_with_reused_index() { + for id in [None, Some("dup")] { + let mut h = Harness::with( + StreamingToolCallTracker::new().with_generate_id(|| "generated-id".to_string()), + ); + + for value in ["{\"value\":1}", "{\"value\":"] { + h.delta(delta( + Some(0), + id, + Some("function"), + Some("same_tool"), + Some(value), + )) + .unwrap(); + } + h.flush(); + + let calls = h.tool_calls(); + let inputs: Vec<&str> = calls.iter().map(|c| c.2.as_str()).collect(); + assert_eq!(inputs, vec!["{\"value\":1}", "{\"value\":"], "id {id:?}"); + let ids: std::collections::HashSet<&str> = calls.iter().map(|c| c.0.as_str()).collect(); + assert_eq!(ids.len(), 2, "id {id:?}"); + } + } + + #[test] + fn keeps_interleaved_same_name_calls_with_distinct_ids_and_reused_index_separate() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "same_tool", "{\"value\":")) + .unwrap(); + h.delta(start(0, "call_2", "same_tool", "{\"value\":2}")) + .unwrap(); + h.delta(delta(Some(0), Some("call_1"), None, None, Some("1}"))) + .unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![ + tc("call_1", "same_tool", "{\"value\":1}"), + tc("call_2", "same_tool", "{\"value\":2}"), + ] + ); + } + + #[test] + fn ignores_an_index_only_continuation_after_the_index_is_reused() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "first", "{\"value\":1}")) + .unwrap(); + h.delta(start(0, "call_2", "second", "{\"value\":2}")) + .unwrap(); + h.delta(cont(Some(0), "{\"unattributed\":true}")).unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![ + tc("call_1", "first", "{\"value\":1}"), + tc("call_2", "second", "{\"value\":2}"), + ] + ); + } + + #[test] + fn uses_the_index_when_continuation_ids_are_blank() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "read_file", "{\"pa")).unwrap(); + h.delta(delta(Some(0), Some(" "), None, None, Some("th\":\"a\"}"))) + .unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![tc("call_1", "read_file", "{\"path\":\"a\"}")] + ); + } + + #[test] + fn generates_unique_ids_for_blank_and_repeated_ids() { + let (generate, _) = counting_generator(&["generated-1", "generated-2"]); + let mut h = Harness::with(StreamingToolCallTracker::new().with_generate_id(generate)); + + h.delta(start(0, "", "read_file", "{}")).unwrap(); + h.delta(start(1, "dup", "read_file", "{}")).unwrap(); + h.delta(start(2, "dup", "write_file", "{}")).unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![ + tc("generated-1", "read_file", "{}"), + tc("dup", "read_file", "{}"), + tc("generated-2", "write_file", "{}"), + ] + ); + } + + #[test] + fn keeps_same_name_calls_with_repeated_ids_and_distinct_indices_separate() { + let mut h = Harness::with( + StreamingToolCallTracker::new().with_generate_id(|| "generated-id".to_string()), + ); + + h.delta(start(0, "dup", "same_tool", "{\"value\":0}")) + .unwrap(); + h.delta(start(1, "dup", "same_tool", "{\"value\":1}")) + .unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![ + tc("dup", "same_tool", "{\"value\":0}"), + tc("generated-id", "same_tool", "{\"value\":1}"), + ] + ); + } + + #[test] + fn preserves_nonblank_ids_and_function_names_exactly() { + let mut h = Harness::new(); + + h.delta(start(0, " spaced ", " same_tool ", "{\"value\":")) + .unwrap(); + h.delta(delta(Some(0), Some(" spaced "), None, None, Some("0}"))) + .unwrap(); + h.delta(start(1, "spaced", " same_tool ", "{\"value\":1}")) + .unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![ + tc(" spaced ", " same_tool ", "{\"value\":0}"), + tc("spaced", " same_tool ", "{\"value\":1}"), + ] + ); + } + + #[test] + fn creates_bounded_unique_ids_when_generate_id_returns_duplicates() { + let (generate, calls) = counting_generator(&["generated-id"]); + let mut h = Harness::with(StreamingToolCallTracker::new().with_generate_id(generate)); + + for (index, name) in [(0, "first"), (1, "second"), (2, "third")] { + h.delta(delta( + Some(index), + None, + Some("function"), + Some(name), + Some("{}"), + )) + .unwrap(); + } + h.flush(); + + assert_eq!(calls.load(Ordering::SeqCst), 3); + let ids: Vec = h.tool_calls().into_iter().map(|c| c.0).collect(); + assert_eq!( + ids, + vec!["generated-id", "generated-id-1", "generated-id-2"] + ); + } + + #[test] + fn creates_usable_ids_when_generate_id_returns_blank_values() { + let (generate, calls) = counting_generator(&[" "]); + let mut h = Harness::with(StreamingToolCallTracker::new().with_generate_id(generate)); + + for (index, name) in [(0, "first"), (1, "second")] { + h.delta(delta( + Some(index), + None, + Some("function"), + Some(name), + Some("{}"), + )) + .unwrap(); + } + h.flush(); + + assert_eq!(calls.load(Ordering::SeqCst), 2); + let ids: Vec = h.tool_calls().into_iter().map(|c| c.0).collect(); + assert_eq!(ids, vec!["tool-call", "tool-call-1"]); + } + + #[test] + fn ignores_unattributable_deltas_when_multiple_calls_are_active() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "read_file", "{\"path\":\"a\"}")) + .unwrap(); + h.delta(start(1, "call_2", "write_file", "{\"path\":\"b\"}")) + .unwrap(); + h.delta(cont(None, "{\"unattributed\":true}")).unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![ + tc("call_1", "read_file", "{\"path\":\"a\"}"), + tc("call_2", "write_file", "{\"path\":\"b\"}"), + ] + ); + } + + #[test] + fn ignores_an_ambiguous_continuation_for_a_repeated_id() { + let mut h = Harness::with( + StreamingToolCallTracker::new().with_generate_id(|| "generated-id".to_string()), + ); + + h.delta(start(0, "dup", "read_file", "{\"path\":\"a\"}")) + .unwrap(); + h.delta(start(1, "dup", "write_file", "{\"path\":\"b\"}")) + .unwrap(); + h.delta(delta( + None, + Some("dup"), + None, + None, + Some("{\"unattributed\":true}"), + )) + .unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![ + tc("dup", "read_file", "{\"path\":\"a\"}"), + tc("generated-id", "write_file", "{\"path\":\"b\"}"), + ] + ); + } + + #[test] + fn uses_a_matching_name_and_index_for_an_id_less_continuation() { + let mut h = Harness::with( + StreamingToolCallTracker::new().with_type_validation(TypeValidation::Required), + ); + + h.delta(start(0, "call_1", "read_file", "{\"pa")).unwrap(); + h.delta(delta( + Some(0), + None, + None, + Some("read_file"), + Some("th\":\"a\"}"), + )) + .unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![tc("call_1", "read_file", "{\"path\":\"a\"}")] + ); + } + + #[test] + fn uses_the_index_when_a_continuation_has_an_unexpected_id() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "read_file", "{\"pa")).unwrap(); + h.delta(delta( + Some(0), + Some("unexpected"), + None, + None, + Some("th\":\"a\"}"), + )) + .unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![tc("call_1", "read_file", "{\"path\":\"a\"}")] + ); + } + + #[test] + fn continues_a_call_when_its_id_changes_but_index_and_name_match() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "read_file", "{\"pa")).unwrap(); + h.delta(start(0, "unexpected", "read_file", "th\":\"a\"}")) + .unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![tc("call_1", "read_file", "{\"path\":\"a\"}")] + ); + } + + #[test] + fn continues_a_call_when_all_labels_repeat_after_a_parsable_argument_prefix() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "calculate", "1")).unwrap(); + h.delta(start(0, "call_1", "calculate", "2")).unwrap(); + h.flush(); + + assert_eq!(h.tool_calls(), vec![tc("call_1", "calculate", "12")]); + } + + #[test] + fn continues_a_structured_argument_when_repeated_labels_precede_a_nested_object() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "calculate", "{\"value\":")) + .unwrap(); + h.delta(start(0, "call_1", "calculate", "{\"nested\":true}}")) + .unwrap(); + h.flush(); + + assert_eq!( + h.tool_calls(), + vec![tc("call_1", "calculate", "{\"value\":{\"nested\":true}}")] + ); + } + + #[test] + fn emits_tool_calls_in_index_order() { + let mut h = Harness::new(); + + h.delta(start(1, "call_1", "second", "{}")).unwrap(); + h.delta(start(0, "call_0", "first", "{}")).unwrap(); + h.flush(); + + let names: Vec = h.tool_calls().into_iter().map(|c| c.1).collect(); + assert_eq!(names, vec!["first", "second"]); + } + + #[test] + fn preserves_insertion_order_when_calls_mix_present_and_omitted_indices() { + let mut h = Harness::new(); + + h.delta(delta( + None, + Some("call_without_index"), + Some("function"), + Some("first"), + Some("{}"), + )) + .unwrap(); + h.delta(start(0, "call_with_index", "second", "{}")) + .unwrap(); + h.flush(); + + let names: Vec = h.tool_calls().into_iter().map(|c| c.1).collect(); + assert_eq!(names, vec!["first", "second"]); + } + + #[test] + fn errors_when_function_name_is_missing_from_a_new_call() { + let mut h = Harness::new(); + + // `function: {}` — a function object with neither name nor arguments. + let result = h.delta(delta(Some(0), Some("call_1"), Some("function"), None, None)); + + assert_eq!(result, Err(TrackerError::MissingFunctionName)); + } +} + +mod type_validation { + use super::*; + + fn custom_type() -> StreamingToolCallDelta { + delta( + Some(0), + Some("call_1"), + Some("custom"), + Some("fn"), + Some(""), + ) + } + + fn no_type() -> StreamingToolCallDelta { + delta(Some(0), Some("call_1"), None, Some("fn"), Some("")) + } + + #[test] + fn does_not_validate_type_with_none() { + let mut h = Harness::with( + StreamingToolCallTracker::new().with_type_validation(TypeValidation::None), + ); + + assert_eq!(h.delta(custom_type()), Ok(())); + } + + #[test] + fn validates_type_when_present_with_if_present() { + let mut h = Harness::with( + StreamingToolCallTracker::new().with_type_validation(TypeValidation::IfPresent), + ); + + assert_eq!(h.delta(custom_type()), Err(TrackerError::InvalidType)); + assert_eq!( + TrackerError::InvalidType.to_string(), + "Expected 'function' type." + ); + + // A missing type is accepted. + assert_eq!(h.delta(no_type()), Ok(())); + } + + #[test] + fn requires_function_type_with_required() { + let mut h = Harness::with( + StreamingToolCallTracker::new().with_type_validation(TypeValidation::Required), + ); + + assert_eq!(h.delta(no_type()), Err(TrackerError::InvalidType)); + + assert_eq!(h.delta(start(0, "call_1", "fn", "")), Ok(())); + } +} + +mod flush { + use super::*; + + #[test] + fn finalizes_unfinished_tool_calls() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "fn", "{\"key\": \"val")) + .unwrap(); + h.clear(); + + h.flush(); + + assert_eq!( + h.parts, + vec![ + input_end("call_1"), + tool_call("call_1", "fn", "{\"key\": \"val"), + ] + ); + } + + #[test] + fn does_not_re_finalize_already_finished_tool_calls() { + let mut h = Harness::new(); + + h.delta(start(0, "call_1", "fn", "{}")).unwrap(); + h.flush(); + h.clear(); + + h.flush(); + + assert_eq!(h.parts, vec![]); + } +} + +mod metadata { + use super::*; + + fn google_tracker() -> StreamingToolCallTracker { + StreamingToolCallTracker::new() + .with_extract_metadata(|delta| { + delta.extra["extra_content"]["google"]["thought_signature"] + .as_str() + .map(|sig| json!({ "thoughtSignature": sig })) + }) + .with_build_provider_metadata(|metadata| { + metadata + .and_then(|m| m.get("thoughtSignature")) + .map(|sig| json!({ "google": { "thoughtSignature": sig } })) + }) + } + + #[test] + fn extracts_and_includes_provider_metadata_in_tool_call_parts() { + let mut h = Harness::with(google_tracker()); + + h.delta( + start(0, "call_1", "fn", "{}") + .extra(json!({ "extra_content": { "google": { "thought_signature": "sig123" } } })), + ) + .unwrap(); + h.flush(); + + let tool_call = h.parts.iter().find(|p| matches!(p, Part::ToolCall { .. })); + assert_eq!( + tool_call, + Some(&Part::ToolCall { + tool_call_id: "call_1".into(), + tool_name: "fn".into(), + input: "{}".into(), + provider_metadata: Some(json!({ "google": { "thoughtSignature": "sig123" } })), + }) + ); + } + + #[test] + fn includes_provider_metadata_for_unfinished_tool_calls_finalized_in_flush() { + let mut h = Harness::with( + StreamingToolCallTracker::new() + .with_extract_metadata(|_| Some(json!({ "custom": { "key": "value" } }))) + .with_build_provider_metadata(|metadata| { + metadata.map(|m| json!({ "provider": m })) + }), + ); + + h.delta(start(0, "call_1", "fn", "{\"incomplete")).unwrap(); + h.clear(); + + h.flush(); + + assert_eq!( + h.parts.last(), + Some(&Part::ToolCall { + tool_call_id: "call_1".into(), + tool_name: "fn".into(), + input: "{\"incomplete".into(), + provider_metadata: Some(json!({ "provider": { "custom": { "key": "value" } } })), + }) + ); + } + + #[test] + fn omits_provider_metadata_when_the_builder_returns_none() { + let mut h = Harness::with( + StreamingToolCallTracker::new() + .with_extract_metadata(|_| None) + .with_build_provider_metadata(|_| None), + ); + + h.delta(start(0, "call_1", "fn", "{}")).unwrap(); + h.flush(); + + let tool_call = h.parts.iter().find(|p| matches!(p, Part::ToolCall { .. })); + assert_eq!(tool_call, Some(&super::tool_call("call_1", "fn", "{}"))); + } +} + +mod generate_id { + use super::*; + + #[test] + fn keeps_the_wire_id_when_present() { + let mut h = Harness::with( + StreamingToolCallTracker::new().with_generate_id(|| "custom-id".to_string()), + ); + + h.delta(start(0, "call_1", "fn", "{\"key\": \"val")) + .unwrap(); + h.clear(); + + h.flush(); + + let ids: Vec = h.tool_calls().into_iter().map(|c| c.0).collect(); + assert_eq!(ids, vec!["call_1"]); + } +} diff --git a/aimux-providers/src/openai/model.rs b/aimux-providers/src/openai/model.rs index 6445d52c..107f984e 100644 --- a/aimux-providers/src/openai/model.rs +++ b/aimux-providers/src/openai/model.rs @@ -20,7 +20,9 @@ use aimux_core::result::{GenerateContent, GenerateResult, StreamResult}; use aimux_core::stream_part::StreamPart; use aimux_core::types::{FinishReason, FinishReasonUnified, ResponseMetadata, Usage}; -use aimux_provider_utils::HttpRequest; +use aimux_provider_utils::{ + HttpRequest, StreamingToolCallDelta, StreamingToolCallFunction, StreamingToolCallTracker, +}; use super::OpenAIConfig; use super::convert::{RequestBodyResult, build_request_body_with_warnings, parse_finish_reason}; @@ -176,15 +178,6 @@ fn convert_usage(usage: &UsageResponse, usage_raw: Option<&Value>) -> Usage { } } -// ── Tool-call accumulator (streaming) ──────────────────────────────────────── - -/// Accumulates a streamed tool call's id, name, and argument fragments. -struct ToolCallAccumulator { - id: String, - name: String, - arguments: String, -} - #[async_trait] impl LanguageModel for OpenAIModel { /// Provider identity for recording/routing. Uses `config.provider` (the @@ -491,9 +484,9 @@ pub async fn execute_stream( let mut response_metadata_emitted = false; let mut final_logprobs: Option = None; - // Tool-call accumulators keyed by OpenAI's `index` field. - let mut tool_calls: HashMap = HashMap::new(); - let mut tool_call_order: Vec = Vec::new(); + // Streamed tool calls, correlated by wire id, index and function name + // (the AI SDK's StreamingToolCallTracker) and finalized on flush. + let mut tool_calls = StreamingToolCallTracker::new(); // Process the first event (already peeked) then the rest. let mut event_iter = @@ -643,49 +636,30 @@ pub async fn execute_stream( reasoning_started = false; } for dtc in tool_call_deltas { - let idx = dtc.index; - let func = dtc.function.unwrap_or_default(); - - // New tool call: has id and/or name. - let is_new = !tool_calls.contains_key(&idx); - if is_new { - let id = dtc.id.unwrap_or_default(); - let name = func.name.unwrap_or_default(); - tool_calls.insert( - idx, - ToolCallAccumulator { - id: id.clone(), - name: name.clone(), - arguments: String::new(), - }, - ); - tool_call_order.push(idx); - yield Ok(StreamPart::ToolInputStart { - id, - tool_name: name, - provider_executed: None, - dynamic: None, - title: None, - provider_metadata: None, - }); - } - - // Argument delta. - // For new tool calls, skip the delta when - // arguments are empty (matches TS — the - // initial `""` is not emitted). For - // continuation chunks, always emit (even - // empty, matching TS). - if let Some(args) = func.arguments - && (!is_new || !args.is_empty()) - && let Some(acc) = tool_calls.get_mut(&idx) { - acc.arguments.push_str(&args); - yield Ok(StreamPart::ToolInputDelta { - id: acc.id.clone(), - delta: args, - provider_metadata: None, - }); + let delta = StreamingToolCallDelta { + index: dtc.index, + id: dtc.id, + r#type: None, + function: dtc.function.map(|f| StreamingToolCallFunction { + name: f.name, + arguments: f.arguments, + }), + extra: Value::Null, + }; + match tool_calls.process_delta(&delta) { + Ok(parts) => { + for part in parts { + yield Ok(part); } + } + // A malformed delta (new call without a + // function name) is invalid response + // data, as in the AI SDK; the stream ends. + Err(error) => { + yield Err(error.into()); + return; + } + } } } @@ -763,27 +737,10 @@ pub async fn execute_stream( }); } - // A parsable argument buffer can still be a prefix of a longer input. - // Match AI SDK's tracker by finalizing only when the stream flushes. - for &idx in &tool_call_order { - if let Some(acc) = tool_calls.get(&idx) { - yield Ok(StreamPart::ToolInputEnd { - id: acc.id.clone(), - provider_metadata: None, - }); - let input = Value::String(acc.arguments.clone()); - yield Ok(StreamPart::ToolCall { - tool_call_id: acc.id.clone(), - tool_name: acc.name.clone(), - input, - provider_executed: None, - dynamic: None, - thought_signature: None, - invalid: None, - error: None, - provider_metadata: None, - }); - } + // A parsable argument buffer can still be a prefix of a longer input: + // like the AI SDK's tracker, finalize only when the stream flushes. + for part in tool_calls.flush() { + yield Ok(part); } // Build provider metadata for the Finish part. diff --git a/aimux-providers/src/openai/types.rs b/aimux-providers/src/openai/types.rs index ea78af6e..1a8c8340 100644 --- a/aimux-providers/src/openai/types.rs +++ b/aimux-providers/src/openai/types.rs @@ -156,8 +156,10 @@ pub struct Delta { #[derive(Debug, Deserialize)] pub struct DeltaToolCall { + /// Optional: some OpenAI-compatible providers omit it (the AI SDK schema + /// is `index: z.number().nullish()`); the tracker falls back to id/name. #[serde(default)] - pub index: usize, + pub index: Option, #[serde(default)] pub id: Option, #[serde(default)] diff --git a/aimux-stream/Cargo.toml b/aimux-stream/Cargo.toml index 6ee65c24..4f167f01 100644 --- a/aimux-stream/Cargo.toml +++ b/aimux-stream/Cargo.toml @@ -4,18 +4,21 @@ version.workspace = true edition.workspace = true license.workspace = true description = "Streaming primitives for aimux" -keywords = ["llm", "streaming", "sse", "ndjson"] +keywords = ["llm", "streaming", "sse"] categories = ["asynchronous", "parser-implementations"] documentation = "https://docs.rs/aimux-stream" [dependencies] futures = { workspace = true } -tokio = { workspace = true } pin-project-lite = { workspace = true } bytes = { workspace = true } -serde = { workspace = true } -serde_json = { workspace = true } thiserror = { workspace = true } +sse-stream = { version = "0.3.0", default-features = false, features = ["memchr"] } + +[dev-dependencies] +serde_json = { workspace = true } +sha2 = "0.10" +tokio = { workspace = true } [lints] workspace = true diff --git a/aimux-stream/src/lib.rs b/aimux-stream/src/lib.rs index 8091dbeb..3edc0c1a 100644 --- a/aimux-stream/src/lib.rs +++ b/aimux-stream/src/lib.rs @@ -1,18 +1,13 @@ //! # aimux-stream //! -//! Low-level streaming primitives for SSE (Server-Sent Events) and NDJSON parsing, -//! used by provider implementations to decode model API response streams. +//! Low-level SSE (Server-Sent Events) decoding, used by +//! `aimux-provider-utils` to decode model API response streams. The crate +//! plays the role `eventsource-parser` plays for the AI SDK: it knows nothing +//! about providers, models or stream parts. pub mod lines; -pub mod ndjson; pub mod sse; -pub mod streaming_tool_call_tracker; // Re-export the most commonly used items. pub use lines::extract_lines; -pub use ndjson::{NdjsonError, NdjsonStream}; pub use sse::{SseError, SseEvent, SseStream}; -pub use streaming_tool_call_tracker::{ - StreamingToolCallDelta, StreamingToolCallFunction, StreamingToolCallTracker, - ToolCallStreamPart, TrackerError, TypeValidation, -}; diff --git a/aimux-stream/src/ndjson.rs b/aimux-stream/src/ndjson.rs deleted file mode 100644 index ca6801a5..00000000 --- a/aimux-stream/src/ndjson.rs +++ /dev/null @@ -1,146 +0,0 @@ -//! NDJSON (newline-delimited JSON) stream decoder. - -use bytes::Bytes; -use futures::Stream; -use pin_project_lite::pin_project; -use serde::de::DeserializeOwned; -use std::pin::Pin; -use std::task::{Context, Poll}; -use thiserror::Error; - -/// Default upper bound on a single buffered NDJSON line's size (1 MiB). -const DEFAULT_MAX_LINE_SIZE: usize = 1024 * 1024; - -#[derive(Debug, Error)] -pub enum NdjsonError { - #[error("utf-8 error: {0}")] - Utf8(#[from] std::string::FromUtf8Error), - #[error("json error: {0}")] - Json(#[from] serde_json::Error), - #[error("stream error: {0}")] - Stream(String), - #[error("NDJSON line exceeded maximum allowed size")] - LineTooLarge, -} - -pin_project! { - /// Decodes a byte stream into parsed NDJSON values. - /// - /// Bytes are accumulated in a raw `Vec` buffer and split on `\n`. Each - /// complete line is strictly UTF-8 decoded only *after* reassembly, so a - /// multi-byte character split across two network chunks is never corrupted - /// into replacement chars — unlike a per-chunk - /// `String::from_utf8_lossy` decode, which would emit a `U+FFFD` on each - /// side of the split. - pub struct NdjsonStream { - #[pin] - inner: S, - buffer: Vec, - max_line_size: usize, - _marker: std::marker::PhantomData<(T, E)>, - } -} - -impl NdjsonStream -where - S: Stream> + Unpin, - T: DeserializeOwned, -{ - pub fn new(stream: S) -> Self { - Self::with_max_line_size(stream, DEFAULT_MAX_LINE_SIZE) - } - - /// Create an [`NdjsonStream`] with a custom per-line size limit. A line - /// longer than `max_line_size` bytes (excluding the `\n`) yields - /// [`NdjsonError::LineTooLarge`], as does a buffer that grows past the - /// limit while waiting for a newline. - pub fn with_max_line_size(stream: S, max_line_size: usize) -> Self { - Self { - inner: stream, - buffer: Vec::new(), - max_line_size, - _marker: std::marker::PhantomData, - } - } -} - -impl Stream for NdjsonStream -where - S: Stream> + Unpin, - T: DeserializeOwned, - E: std::fmt::Display, -{ - type Item = Result; - - fn poll_next(mut self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll> { - let this = self.as_mut().get_mut(); - - loop { - // Try to parse a complete JSON line. - if let Some(pos) = this.buffer.iter().position(|&b| b == b'\n') { - if pos > this.max_line_size { - // Drop the oversized line so a retried poll makes progress. - this.buffer.drain(..pos + 1); - return Poll::Ready(Some(Err(NdjsonError::LineTooLarge))); - } - // Extract the line bytes (without the newline) and decode - // strictly. Decoding the fully reassembled line (rather than - // `from_utf8_lossy` per chunk) preserves code points split - // across chunks and surfaces invalid UTF-8 as an error. - let line_bytes: Vec = this.buffer.drain(..pos).collect(); - this.buffer.drain(..1); // the newline - let line = match String::from_utf8(line_bytes) { - Ok(s) => s, - Err(e) => return Poll::Ready(Some(Err(NdjsonError::Utf8(e)))), - }; - let line = line.trim(); - if line.is_empty() { - continue; - } - match serde_json::from_str::(line) { - Ok(value) => return Poll::Ready(Some(Ok(value))), - Err(e) => return Poll::Ready(Some(Err(NdjsonError::Json(e)))), - } - } - - // No complete line yet. Guard against unbounded buffer growth when - // a newline never arrives. The partial can never form a valid line - // under the limit, so drop it — a retried poll then makes progress - // instead of looping on the same oversized buffer. - if this.buffer.len() > this.max_line_size { - this.buffer.clear(); - return Poll::Ready(Some(Err(NdjsonError::LineTooLarge))); - } - - // Read more bytes. - match Pin::new(&mut this.inner).poll_next(cx) { - Poll::Ready(Some(Ok(bytes))) => { - this.buffer.extend_from_slice(&bytes); - } - Poll::Ready(Some(Err(e))) => { - return Poll::Ready(Some(Err(NdjsonError::Stream(e.to_string())))); - } - Poll::Ready(None) => { - // Stream ended — try to parse any remaining buffer. - if this.buffer.len() > this.max_line_size { - return Poll::Ready(Some(Err(NdjsonError::LineTooLarge))); - } - let remaining_bytes = std::mem::take(&mut this.buffer); - let remaining = match String::from_utf8(remaining_bytes) { - Ok(s) => s, - Err(e) => return Poll::Ready(Some(Err(NdjsonError::Utf8(e)))), - }; - let remaining = remaining.trim(); - if remaining.is_empty() { - return Poll::Ready(None); - } - match serde_json::from_str::(remaining) { - Ok(value) => return Poll::Ready(Some(Ok(value))), - Err(e) => return Poll::Ready(Some(Err(NdjsonError::Json(e)))), - } - } - Poll::Pending => return Poll::Pending, - } - } - } -} diff --git a/aimux-stream/src/sse.rs b/aimux-stream/src/sse.rs index c53f66e4..f32f436c 100644 --- a/aimux-stream/src/sse.rs +++ b/aimux-stream/src/sse.rs @@ -1,249 +1,124 @@ -//! SSE (Server-Sent Events) parser for streaming model responses. +//! SSE (Server-Sent Events) decoding for streaming model responses. +//! +//! The parser is the [`sse-stream`](https://crates.io/crates/sse-stream) crate +//! (the WHATWG "parsing an event stream" algorithm: `\n`, `\r` and `\r\n` +//! line endings in any mix, a leading UTF-8 BOM, `field: value` with one +//! optional leading space, lines without `:` as empty-valued fields, +//! `:`-prefixed comments, an `id` containing U+0000 ignored, `retry` only when +//! all ASCII digits, unknown fields ignored, a partial block at end of stream +//! dropped). +//! +//! This module is the thin adapter that gives it the AI SDK's dispatch +//! semantics, the way `parseJsonEventStream` wraps `eventsource-parser`: +//! +//! - a block is dispatched only if it had at least one `data` line; comment- +//! only and metadata-only blocks are dropped; +//! - an empty `event` value means no event type; +//! - `retry` is reported on the event it was parsed with (upstream reports it +//! through `onRetry`), so a `retry` block without `data` is dropped; +//! - there is no buffer size limit, as in `parseJsonEventStream`. +//! +//! Differences from upstream kept on purpose: a field value that is not valid +//! UTF-8 is a strict [`SseError::Utf8`] and ends the stream (upstream decodes +//! lossily), and a transport error ends the stream after it is reported. + +use std::marker::PhantomData; +use std::pin::Pin; +use std::task::{Context, Poll, ready}; use bytes::Bytes; use futures::Stream; use pin_project_lite::pin_project; -use std::pin::Pin; -use std::task::{Context, Poll}; +use sse_stream::SseByteStream; use thiserror::Error; -/// Default upper bound on a single buffered SSE event's size (1 MiB). -const DEFAULT_MAX_EVENT_SIZE: usize = 1024 * 1024; - +/// A terminal decoding error: after yielding one, the stream ends. #[derive(Debug, Error)] pub enum SseError { + /// A recognized field value (`data`, `event`, `id`, `retry`) is not valid + /// UTF-8. #[error("utf-8 decode error: {0}")] - Utf8(#[from] std::string::FromUtf8Error), + Utf8(#[source] std::str::Utf8Error), + /// The underlying byte stream failed; the source is the transport error. #[error("stream error: {0}")] - Stream(String), - #[error("SSE frame exceeded maximum allowed size")] - FrameTooLarge, + Stream(#[source] Box), + /// Any other decoder error. + #[error("SSE decode error: {0}")] + Decode(#[source] sse_stream::Error), +} + +impl From for SseError { + fn from(error: sse_stream::Error) -> Self { + match error { + sse_stream::Error::Body(source) => Self::Stream(source), + sse_stream::Error::Utf8Parse(source) => Self::Utf8(source), + other => Self::Decode(other), + } + } } /// A parsed SSE event. #[derive(Debug, Clone, Default)] pub struct SseEvent { - /// The `event:` field (optional). + /// The `event:` field (`None` when absent or empty). pub event: Option, - /// The `data:` field. + /// The `data:` field; multiple `data` lines are joined with `\n`. pub data: String, /// The `id:` field (optional). pub id: Option, - /// The `retry:` field (optional). + /// The `retry:` field (optional; all-digit values only). pub retry: Option, } pin_project! { /// An adapter that decodes a byte stream into SSE events. - /// - /// Bytes are accumulated in a raw `Vec` buffer and split on the SSE - /// event terminator (a blank line). Each complete frame is strictly - /// UTF-8 decoded only *after* reassembly, so a multi-byte character split - /// across two network chunks is never corrupted into replacement chars — - /// unlike a per-chunk `String::from_utf8_lossy` decode, which would emit a - /// `U+FFFD` on each side of the split. - pub struct SseStream { + pub struct SseStream + where + S: Stream>, + { #[pin] - inner: S, - buffer: Vec, - max_event_size: usize, - done: bool, - _err: std::marker::PhantomData, + inner: SseByteStream, + _err: PhantomData, } } impl SseStream where - S: Stream> + Unpin, + S: Stream>, { pub fn new(stream: S) -> Self { - Self::with_max_event_size(stream, DEFAULT_MAX_EVENT_SIZE) - } - - /// Create an [`SseStream`] with a custom per-event size limit. A single - /// event frame (the bytes between two terminators, excluding the - /// terminator itself) larger than `max_event_size` bytes yields - /// [`SseError::FrameTooLarge`], as does a buffer that grows past the limit - /// while waiting for a terminator. - pub fn with_max_event_size(stream: S, max_event_size: usize) -> Self { Self { - inner: stream, - buffer: Vec::new(), - max_event_size, - done: false, - _err: std::marker::PhantomData, + inner: SseByteStream::new(stream), + _err: PhantomData, } } } impl Stream for SseStream where - S: Stream> + Unpin, - E: std::fmt::Display, + S: Stream>, + E: std::error::Error + Send + Sync + 'static, { type Item = Result; - fn poll_next(mut self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll> { - let this = self.as_mut().get_mut(); - + fn poll_next(self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll> { + let mut this = self.project(); loop { - // Try to emit a complete event from the buffer. - if let Some((frame_len, sep_len)) = find_separator(&this.buffer) { - if frame_len > this.max_event_size { - // Drop the oversized frame so a retried poll makes progress. - this.buffer.drain(..frame_len + sep_len); - return Poll::Ready(Some(Err(SseError::FrameTooLarge))); - } - // Extract the frame bytes and drop the terminator. Decoding the - // fully reassembled frame with `String::from_utf8` (rather than - // `from_utf8_lossy` per chunk) preserves code points split - // across chunks and surfaces invalid UTF-8 as an error. - let frame_bytes: Vec = this.buffer.drain(..frame_len).collect(); - this.buffer.drain(..sep_len); - let frame = match String::from_utf8(frame_bytes) { - Ok(s) => s, - Err(e) => return Poll::Ready(Some(Err(SseError::Utf8(e)))), - }; - let (event, has_data_line) = try_parse_event(&frame); - // Per the SSE spec / eventsource-parser, an event is dispatched - // only if it had at least one `data:` line (dataLines > 0). - // This covers comment-only, `event:`/`id:`/`retry:`-only, and - // blank-line keep-alives (all dispatched as nothing), while an - // explicit empty data line (`data:\n\n`) is still dispatched. - if !has_data_line { - continue; + match ready!(this.inner.as_mut().poll_next(cx)) { + None => return Poll::Ready(None), + Some(Err(error)) => return Poll::Ready(Some(Err(error.into()))), + Some(Ok(block)) => { + // Dispatch only blocks that carried at least one `data` + // line (`dataLines > 0` upstream). + let Some(data) = block.data else { continue }; + return Poll::Ready(Some(Ok(SseEvent { + event: block.event.filter(|event| !event.is_empty()), + data, + id: block.id, + retry: block.retry, + }))); } - return Poll::Ready(Some(Ok(event))); - } - - // No complete event yet. Guard against unbounded buffer growth when - // a terminator never arrives. The partial can never form a valid - // frame under the limit, so drop it — a retried poll then makes - // progress instead of looping on the same oversized buffer. - if this.buffer.len() > this.max_event_size { - this.buffer.clear(); - return Poll::Ready(Some(Err(SseError::FrameTooLarge))); - } - - if this.done { - // No terminating blank line: the buffered partial is not a - // complete event and is dropped. This matches eventsource-parser's - // `EventSourceParserStream`, which has no flush handler — a - // partial event at EOF is simply not dispatched. - return Poll::Ready(None); - } - - // Read more data. - match Pin::new(&mut this.inner).poll_next(cx) { - Poll::Ready(Some(Ok(bytes))) => { - this.buffer.extend_from_slice(&bytes); - } - Poll::Ready(Some(Err(e))) => { - return Poll::Ready(Some(Err(SseError::Stream(e.to_string())))); - } - Poll::Ready(None) => { - this.done = true; - } - Poll::Pending => return Poll::Pending, } } } } - -/// Locate the next SSE event terminator (a blank line) in `buf`. -/// -/// Returns `(frame_len, sep_len)` where `frame_len` is the number of bytes -/// *before* the terminator and `sep_len` is the terminator's length. Prefers -/// `\n\n` (so a pure-CRLF stream — which contains no bare `\n\n` — falls -/// through to `\r\n\r\n`), mirroring the str-based -/// `find("\n\n").or_else(|| find("\r\n\r\n"))`. -fn find_separator(buf: &[u8]) -> Option<(usize, usize)> { - if let Some(pos) = find_subsequence(buf, b"\n\n") { - return Some((pos, 2)); - } - find_subsequence(buf, b"\r\n\r\n").map(|pos| (pos, 4)) -} - -/// First index of `needle` in `haystack`, comparing raw bytes. -fn find_subsequence(haystack: &[u8], needle: &[u8]) -> Option { - haystack.windows(needle.len()).position(|w| w == needle) -} - -/// Parse a single complete SSE event frame (the bytes between two -/// terminators, already strictly UTF-8 decoded) into an [`SseEvent`] plus a -/// flag indicating whether the event contained at least one `data:` line. -/// -/// A complete event is terminated by a blank line: `\n\n` or `\r\n\r\n` (the -/// terminator is consumed by the caller before this runs). Per the SSE spec, -/// exactly one leading U+0020 SPACE after the `:` is removed from each field -/// value (not all leading whitespace). Comment lines (starting with `:`) and -/// unknown fields are ignored. -fn try_parse_event(frame: &str) -> (SseEvent, bool) { - let mut event = SseEvent::default(); - let mut has_data_line = false; - for line in frame.lines() { - if let Some(value) = field_value(line, "data:") { - has_data_line = true; - if event.data.is_empty() { - event.data = value.to_string(); - } else { - event.data.push('\n'); - event.data.push_str(value); - } - } else if let Some(value) = field_value(line, "event:") { - event.event = Some(value.to_string()); - } else if let Some(value) = field_value(line, "id:") { - event.id = Some(value.to_string()); - } else if let Some(value) = field_value(line, "retry:") { - event.retry = value.parse().ok(); - } - // Comment lines (`:` prefix) and unknown fields are ignored. - } - - (event, has_data_line) -} - -/// Strip `prefix` from `line` and remove exactly one leading U+0020 SPACE from -/// the remainder (per the SSE spec). Returns `None` if `line` does not start -/// with `prefix`. -fn field_value<'a>(line: &'a str, prefix: &str) -> Option<&'a str> { - let rest = line.strip_prefix(prefix)?; - // Remove exactly one leading space, if present. - Some(rest.strip_prefix(' ').unwrap_or(rest)) -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn parse_single_event() { - let (event, has_data_line) = try_parse_event("data: hello world"); - assert_eq!(event.data, "hello world"); - assert!(has_data_line); - } - - #[test] - fn parse_multi_line_data() { - let (event, has_data_line) = try_parse_event("data: line1\ndata: line2"); - assert_eq!(event.data, "line1\nline2"); - assert!(has_data_line); - } - - #[test] - fn parse_event_with_type() { - let (event, has_data_line) = try_parse_event("event: message\ndata: payload"); - assert_eq!(event.event.as_deref(), Some("message")); - assert_eq!(event.data, "payload"); - assert!(has_data_line); - } - - #[test] - fn parse_event_without_data_has_no_data_line() { - // An `event:`-only event has no data line -> not dispatched. - let (event, has_data_line) = try_parse_event("event: ping"); - assert_eq!(event.event.as_deref(), Some("ping")); - assert!(event.data.is_empty()); - assert!(!has_data_line); - } -} diff --git a/aimux-stream/src/streaming_tool_call_tracker.rs b/aimux-stream/src/streaming_tool_call_tracker.rs deleted file mode 100644 index be9ad890..00000000 --- a/aimux-stream/src/streaming_tool_call_tracker.rs +++ /dev/null @@ -1,429 +0,0 @@ -//! Streaming tool call tracker. -//! -//! Rust translation of `@ai-sdk/provider-utils`'s `StreamingToolCallTracker` -//! (`packages/provider-utils/src/streaming-tool-call-tracker.ts`). -//! -//! Tracks streaming tool call state across multiple deltas from an -//! OpenAI-compatible chat completion stream: accumulates `arguments` string -//! fragments by `index`, emits `tool-input-start` / `tool-input-delta` / -//! `tool-input-end` / `tool-call` events, and finalizes any unfinished tool -//! calls on [`StreamingToolCallTracker::flush`]. -//! -//! Like the TS original, a tool call is *never* finalized before `flush` — a -//! parsable argument buffer can still be the prefix of a longer argument -//! string, so acting on it early would use truncated inputs (ai-sdk #13137). - -use serde_json::Value; -use thiserror::Error; - -/// Default upper bound accepted for a tool call `index`. -const DEFAULT_MAX_INDEX: usize = 1024; - -/// The `function` sub-object of a streaming tool call delta. -#[derive(Debug, Clone, Default)] -pub struct StreamingToolCallFunction { - pub name: Option, - pub arguments: Option, -} - -/// A streaming tool call delta — the `tool_calls[i]` entry of an OpenAI-style -/// streaming chunk. -/// -/// Use the builder methods ([`StreamingToolCallDelta::index`], etc.) to -/// construct one ergonomically. `arguments: null` (TS) maps to `None`; an -/// empty-string `arguments: ''` maps to `Some("")`. -#[derive(Debug, Clone, Default)] -pub struct StreamingToolCallDelta { - pub index: Option, - pub id: Option, - /// The `type` field. Named `r#type` because `type` is a reserved word. - pub r#type: Option, - pub function: Option, - /// Provider-specific payload carried alongside the standard fields. Used by - /// `extract_metadata` to pull out provider metadata (e.g. a Google thought - /// signature). Defaults to [`Value::Null`]. - pub extra: Value, -} - -impl StreamingToolCallDelta { - #[must_use] - pub fn new() -> Self { - Self::default() - } - - #[must_use] - pub fn index(mut self, index: usize) -> Self { - self.index = Some(index); - self - } - - #[must_use] - pub fn id(mut self, id: impl Into) -> Self { - self.id = Some(id.into()); - self - } - - /// Set the `type` field (named `tool_type` because `type` is reserved). - #[must_use] - pub fn tool_type(mut self, t: impl Into) -> Self { - self.r#type = Some(t.into()); - self - } - - #[must_use] - pub fn function_name(mut self, name: impl Into) -> Self { - self.function.get_or_insert_with(Default::default).name = Some(name.into()); - self - } - - /// Set the `function.arguments` fragment. Pass `""` for an explicit empty - /// fragment; omit the call entirely for `None` (TS `arguments: null`). - #[must_use] - pub fn arguments(mut self, args: impl Into) -> Self { - self.function.get_or_insert_with(Default::default).arguments = Some(args.into()); - self - } - - #[must_use] - pub fn extra(mut self, extra: Value) -> Self { - self.extra = extra; - self - } -} - -/// The stream parts emitted by [`StreamingToolCallTracker`]. -/// -/// Mirrors the subset of `LanguageModelV4StreamPart` the TS tracker enqueues. -/// Note that [`ToolCallStreamPart::ToolCall::input`] is the raw accumulated -/// argument *string* — the tracker does not parse it (matching the TS -/// behavior). -#[derive(Debug, Clone, PartialEq)] -pub enum ToolCallStreamPart { - /// Start of a tool call's input streaming. - ToolInputStart { id: String, tool_name: String }, - /// A delta of tool call input (a partial argument fragment). - ToolInputDelta { id: String, delta: String }, - /// End of a tool call's input streaming. - ToolInputEnd { id: String }, - /// A complete, finalized tool call. - ToolCall { - tool_call_id: String, - tool_name: String, - input: String, - provider_metadata: Option, - }, -} - -/// How to validate the `type` field on a new tool call delta. -#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] -pub enum TypeValidation { - /// No validation (default). - #[default] - None, - /// Throw if `type` is present and not `"function"`. - IfPresent, - /// Throw if `type` is not exactly `"function"`. - Required, -} - -/// Errors raised while processing a tool call delta. -#[derive(Debug, Error, PartialEq)] -pub enum TrackerError { - #[error("Expected 'id' to be a string.")] - MissingId, - #[error("Expected 'function.name' to be a string.")] - MissingFunctionName, - #[error("Expected 'function' type.")] - InvalidType, - #[error("Tool call index out of range")] - IndexOutOfRange, -} - -struct TrackedToolCall { - id: String, - function_name: String, - arguments: String, - has_finished: bool, - metadata: Option, -} - -/// Extract provider metadata from a delta (the TS `extractMetadata` option). -type ExtractMetadataFn = Box Option>; -/// Build the `providerMetadata` for a finalized tool call (the TS -/// `buildToolCallProviderMetadata` option). -type BuildMetadataFn = Box) -> Option>; - -/// Tracks streaming tool call state across multiple deltas from an -/// OpenAI-compatible chat completion stream. -/// -/// Emitted [`ToolCallStreamPart`]s accumulate in an internal buffer; inspect -/// them with [`parts`](Self::parts) and reset between checks with -/// [`clear_parts`](Self::clear_parts) (the TS test's `parts.length = 0`). -/// -/// The type parameter `M` is the provider-metadata type. Use `()` (the -/// default) when no metadata handling is needed, or `serde_json::Value` (or -/// any `SharedV4ProviderMetadata`-like type) together with -/// [`with_extract_metadata`](Self::with_extract_metadata) / -/// [`with_build_provider_metadata`](Self::with_build_provider_metadata). -pub struct StreamingToolCallTracker { - tool_calls: Vec>>, - parts: Vec>, - // The TS `generateId` option: fallback for `toolCall.id ?? generateId()` - // when an incoming tool-call delta omits its id. - generate_id: Box String>, - type_validation: TypeValidation, - /// Upper bound accepted for a tool call `index`. Guards against a remote - /// index resizing `tool_calls` to a huge vector. Defaults to - /// [`DEFAULT_MAX_INDEX`] (1024). - max_index: usize, - extract_metadata: Option>, - build_provider_metadata: Option>, -} - -impl Default for StreamingToolCallTracker<()> { - fn default() -> Self { - Self::new() - } -} - -impl StreamingToolCallTracker { - /// Create a new tracker with no metadata handling and default settings. - #[must_use] - pub fn new() -> Self { - Self { - tool_calls: Vec::new(), - parts: Vec::new(), - generate_id: Box::new(|| "generated".to_string()), - type_validation: TypeValidation::None, - max_index: DEFAULT_MAX_INDEX, - extract_metadata: None, - build_provider_metadata: None, - } - } - - /// Set a custom id generator (the TS `generateId` option). - #[must_use] - pub fn with_generate_id String + 'static>(mut self, f: F) -> Self { - self.generate_id = Box::new(f); - self - } - - /// Set the `type` validation mode (the TS `typeValidation` option). - #[must_use] - pub fn with_type_validation(mut self, v: TypeValidation) -> Self { - self.type_validation = v; - self - } - - /// Set the maximum accepted tool call `index` (defaults to 1024). A delta - /// whose resolved `index` exceeds `max_index` returns - /// [`TrackerError::IndexOutOfRange`] instead of resizing `tool_calls` to - /// `index + 1` slots. - #[must_use] - pub fn with_max_index(mut self, max_index: usize) -> Self { - self.max_index = max_index; - self - } - - /// Set the metadata extractor (the TS `extractMetadata` option). Called - /// once when a new tool call is detected; the returned metadata is stored - /// on the tool call and passed to the builder at finalization. - #[must_use] - pub fn with_extract_metadata Option + 'static>( - mut self, - f: F, - ) -> Self { - self.extract_metadata = Some(Box::new(f)); - self - } - - /// Set the provider-metadata builder (the TS `buildToolCallProviderMetadata` - /// option). Receives the metadata previously extracted; if `None` is - /// returned, no `provider_metadata` is included in the `tool-call` event. - #[must_use] - pub fn with_build_provider_metadata) -> Option + 'static>( - mut self, - f: F, - ) -> Self { - self.build_provider_metadata = Some(Box::new(f)); - self - } - - /// Process a tool call delta from a streaming response chunk. Emits events - /// into the internal buffer. - /// - /// # Errors - /// - /// Returns `TrackerError::IndexOutOfRange` when the delta's tool index - /// exceeds the configured maximum. - pub fn process_delta(&mut self, delta: &StreamingToolCallDelta) -> Result<(), TrackerError> { - let index = delta.index.unwrap_or(self.tool_calls.len()); - // Guard against a remote `index` resizing `tool_calls` to a huge vector. - if index > self.max_index { - return Err(TrackerError::IndexOutOfRange); - } - let is_new = self - .tool_calls - .get(index) - .map(std::option::Option::is_none) - .unwrap_or(true); - if is_new { - self.process_new_tool_call(index, delta)?; - } else { - self.process_existing_tool_call(index, delta); - } - Ok(()) - } - - /// Finalize any unfinished tool calls. Should be called during the - /// stream's flush to ensure all tool calls are properly completed. Emits - /// `tool-input-end` and `tool-call` for each unfinished tool call. - pub fn flush(&mut self) { - for i in 0..self.tool_calls.len() { - let needs_finish = self - .tool_calls - .get(i) - .and_then(|o| o.as_ref()) - .is_some_and(|tc| !tc.has_finished); - if needs_finish { - self.finish_tool_call_at(i); - } - } - } - - /// The events emitted so far (accumulated across `process_delta`/`flush`). - #[must_use] - pub fn parts(&self) -> &[ToolCallStreamPart] { - &self.parts - } - - /// Clear the accumulated events (the TS test's `parts.length = 0`). - pub fn clear_parts(&mut self) { - self.parts.clear(); - } - - fn process_new_tool_call( - &mut self, - index: usize, - delta: &StreamingToolCallDelta, - ) -> Result<(), TrackerError> { - match self.type_validation { - TypeValidation::Required => { - if delta.r#type.as_deref() != Some("function") { - return Err(TrackerError::InvalidType); - } - } - TypeValidation::IfPresent => { - if delta.r#type.as_deref().is_some_and(|t| t != "function") { - return Err(TrackerError::InvalidType); - } - } - TypeValidation::None => {} - } - - let id = delta.id.clone().unwrap_or_else(|| (self.generate_id)()); - let function_name = delta - .function - .as_ref() - .and_then(|f| f.name.clone()) - .ok_or(TrackerError::MissingFunctionName)?; - - self.parts.push(ToolCallStreamPart::ToolInputStart { - id: id.clone(), - tool_name: function_name.clone(), - }); - - let metadata = self - .extract_metadata - .as_ref() - .and_then(|extract| extract(delta)); - - // TS: `toolCallDelta.function.arguments ?? ''`. - let arguments = delta - .function - .as_ref() - .and_then(|f| f.arguments.clone()) - .unwrap_or_default(); - - if index >= self.tool_calls.len() { - self.tool_calls.resize_with(index + 1, || None); - } - self.tool_calls[index] = Some(TrackedToolCall { - id: id.clone(), - function_name: function_name.clone(), - arguments: arguments.clone(), - has_finished: false, - metadata, - }); - - // Emit initial delta if arguments already present. - if !arguments.is_empty() { - self.parts.push(ToolCallStreamPart::ToolInputDelta { - id: id.clone(), - delta: arguments, - }); - } - - // Tool calls must not finalize before the stream ends (see #13137). - Ok(()) - } - - fn process_existing_tool_call(&mut self, index: usize, delta: &StreamingToolCallDelta) { - // TS: `toolCallDelta.function?.arguments != null`. - let new_args = match delta.function.as_ref().and_then(|f| f.arguments.as_ref()) { - Some(args) => args.clone(), - None => return, - }; - let id = { - let Some(Some(tool_call)) = self.tool_calls.get_mut(index) else { - return; - }; - if tool_call.has_finished { - return; - } - tool_call.arguments.push_str(&new_args); - tool_call.id.clone() - }; - self.parts.push(ToolCallStreamPart::ToolInputDelta { - id, - delta: new_args, - }); - } - - fn finish_tool_call_at(&mut self, index: usize) { - // Scope the mutable borrow of the tracked tool call, extracting the - // owned data we need to emit the final events. - let (id, function_name, arguments, metadata) = { - let Some(Some(tool_call)) = self.tool_calls.get_mut(index) else { - return; - }; - if tool_call.has_finished { - return; - } - ( - tool_call.id.clone(), - tool_call.function_name.clone(), - tool_call.arguments.clone(), - tool_call.metadata.take(), - ) - }; - if let Some(Some(tool_call)) = self.tool_calls.get_mut(index) { - tool_call.has_finished = true; - } - - self.parts - .push(ToolCallStreamPart::ToolInputEnd { id: id.clone() }); - - let provider_metadata = self - .build_provider_metadata - .as_ref() - .and_then(|build| build(metadata.as_ref())); - - self.parts.push(ToolCallStreamPart::ToolCall { - tool_call_id: id, - tool_name: function_name, - input: arguments, - provider_metadata, - }); - } -} diff --git a/aimux-stream/tests/fixtures/eventsource_parser_multibyte.json b/aimux-stream/tests/fixtures/eventsource_parser_multibyte.json new file mode 100644 index 00000000..f427553c --- /dev/null +++ b/aimux-stream/tests/fixtures/eventsource_parser_multibyte.json @@ -0,0 +1 @@ +{"lines":["Blåbærsyltetøy må da være mulig å oppdrive?","በእርግጥ አንድ ሰው ብሉቤሪ ጃምን መግዛት መቻል አለበት?","بالتأكيد يجب أن يكون المرء قادرا على شراء مربى التوت؟","一定能买到蓝莓果酱吗?","ચોક્કસ એક બ્લુબેરી જામ મેળવવા માટે સમર્થ હોવા જ જોઈએ?","בטח אפשר להשיג ריבת אוכמניות?","निश्चित रूप से किसी को ब्लूबेरी जैम खरीदने में सक्षम होना चाहिए?","Bê guman pêdivî ye ku meriv karibe jama şînê peyda bike?","ប្រាកដណាស់ មនុស្សម្នាក់ត្រូវតែអាចទិញយៈសាពូនមី blueberry បានទេ?","ಖಂಡಿತವಾಗಿಯೂ ಒಬ್ಬರು ಬ್ಲೂಬೆರ್ರಿ ಜಾಮ್ ಅನ್ನು ಸಂಗ್ರಹಿಸಲು ಶಕ್ತರಾಗಿರಬೇಕು?","確かにブルーベリージャムを調達できなければなりませんか?","Невже треба вміти добути варення з чорниці?","แน่นอนว่าจะต้องสามารถจัดหาแยมบลูเบอร์รี่ได้?","Сигурно некој мора да може да набави џем од боровинки?"],"emojis":["😶‍🌫️","😮‍💨","😵‍💫","❤️‍🔥","❤️‍🩹","👁️‍🗨️","🫱🏻‍🫲🏼","🫱🏻‍🫲🏽","🫱🏻‍🫲🏾","🫱🏻‍🫲🏿","🫱🏼‍🫲🏻","🫱🏼‍🫲🏽","🫱🏼‍🫲🏾","🫱🏼‍🫲🏿","🫱🏽‍🫲🏻","🫱🏽‍🫲🏼","🫱🏽‍🫲🏾","🫱🏽‍🫲🏿","🫱🏾‍🫲🏻","🫱🏾‍🫲🏼","🫱🏾‍🫲🏽","🫱🏾‍🫲🏿","🫱🏿‍🫲🏻","🫱🏿‍🫲🏼","🫱🏿‍🫲🏽","🫱🏿‍🫲🏾","🧔‍♂️","🧔🏻‍♂️","🧔🏼‍♂️","🧔🏽‍♂️","🧔🏾‍♂️","🧔🏿‍♂️","🧔‍♀️","🧔🏻‍♀️","🧔🏼‍♀️","🧔🏽‍♀️","🧔🏾‍♀️","🧔🏿‍♀️","👨‍🦰","👨🏻‍🦰","👨🏼‍🦰","👨🏽‍🦰","👨🏾‍🦰","👨🏿‍🦰","👨‍🦱","👨🏻‍🦱","👨🏼‍🦱","👨🏽‍🦱","👨🏾‍🦱","👨🏿‍🦱","👨‍🦳","👨🏻‍🦳","👨🏼‍🦳","👨🏽‍🦳","👨🏾‍🦳","👨🏿‍🦳","👨‍🦲","👨🏻‍🦲","👨🏼‍🦲","👨🏽‍🦲","👨🏾‍🦲","👨🏿‍🦲","👩‍🦰","👩🏻‍🦰","👩🏼‍🦰","👩🏽‍🦰","👩🏾‍🦰","👩🏿‍🦰","🧑‍🦰","🧑🏻‍🦰","🧑🏼‍🦰","🧑🏽‍🦰","🧑🏾‍🦰","🧑🏿‍🦰","👩‍🦱","👩🏻‍🦱","👩🏼‍🦱","👩🏽‍🦱","👩🏾‍🦱","👩🏿‍🦱","🧑‍🦱","🧑🏻‍🦱","🧑🏼‍🦱","🧑🏽‍🦱","🧑🏾‍🦱","🧑🏿‍🦱","👩‍🦳","👩🏻‍🦳","👩🏼‍🦳","👩🏽‍🦳","👩🏾‍🦳","👩🏿‍🦳","🧑‍🦳","🧑🏻‍🦳","🧑🏼‍🦳","🧑🏽‍🦳","🧑🏾‍🦳","🧑🏿‍🦳","👩‍🦲","👩🏻‍🦲","👩🏼‍🦲","👩🏽‍🦲","👩🏾‍🦲","👩🏿‍🦲","🧑‍🦲","🧑🏻‍🦲","🧑🏼‍🦲","🧑🏽‍🦲","🧑🏾‍🦲","🧑🏿‍🦲","👱‍♀️","👱🏻‍♀️","👱🏼‍♀️","👱🏽‍♀️","👱🏾‍♀️","👱🏿‍♀️","👱‍♂️","👱🏻‍♂️","👱🏼‍♂️","👱🏽‍♂️","👱🏾‍♂️","👱🏿‍♂️","🙍‍♂️","🙍🏻‍♂️","🙍🏼‍♂️","🙍🏽‍♂️","🙍🏾‍♂️","🙍🏿‍♂️","🙍‍♀️","🙍🏻‍♀️","🙍🏼‍♀️","🙍🏽‍♀️","🙍🏾‍♀️","🙍🏿‍♀️","🙎‍♂️","🙎🏻‍♂️","🙎🏼‍♂️","🙎🏽‍♂️","🙎🏾‍♂️","🙎🏿‍♂️","🙎‍♀️","🙎🏻‍♀️","🙎🏼‍♀️","🙎🏽‍♀️","🙎🏾‍♀️","🙎🏿‍♀️","🙅‍♂️","🙅🏻‍♂️","🙅🏼‍♂️","🙅🏽‍♂️","🙅🏾‍♂️","🙅🏿‍♂️","🙅‍♀️","🙅🏻‍♀️","🙅🏼‍♀️","🙅🏽‍♀️","🙅🏾‍♀️","🙅🏿‍♀️","🙆‍♂️","🙆🏻‍♂️","🙆🏼‍♂️","🙆🏽‍♂️","🙆🏾‍♂️","🙆🏿‍♂️","🙆‍♀️","🙆🏻‍♀️","🙆🏼‍♀️","🙆🏽‍♀️","🙆🏾‍♀️","🙆🏿‍♀️","💁‍♂️","💁🏻‍♂️","💁🏼‍♂️","💁🏽‍♂️","💁🏾‍♂️","💁🏿‍♂️","💁‍♀️","💁🏻‍♀️","💁🏼‍♀️","💁🏽‍♀️","💁🏾‍♀️","💁🏿‍♀️","🙋‍♂️","🙋🏻‍♂️","🙋🏼‍♂️","🙋🏽‍♂️","🙋🏾‍♂️","🙋🏿‍♂️","🙋‍♀️","🙋🏻‍♀️","🙋🏼‍♀️","🙋🏽‍♀️","🙋🏾‍♀️","🙋🏿‍♀️","🧏‍♂️","🧏🏻‍♂️","🧏🏼‍♂️","🧏🏽‍♂️","🧏🏾‍♂️","🧏🏿‍♂️","🧏‍♀️","🧏🏻‍♀️","🧏🏼‍♀️","🧏🏽‍♀️","🧏🏾‍♀️","🧏🏿‍♀️","🙇‍♂️","🙇🏻‍♂️","🙇🏼‍♂️","🙇🏽‍♂️","🙇🏾‍♂️","🙇🏿‍♂️","🙇‍♀️","🙇🏻‍♀️","🙇🏼‍♀️","🙇🏽‍♀️","🙇🏾‍♀️","🙇🏿‍♀️","🤦‍♂️","🤦🏻‍♂️","🤦🏼‍♂️","🤦🏽‍♂️","🤦🏾‍♂️","🤦🏿‍♂️","🤦‍♀️","🤦🏻‍♀️","🤦🏼‍♀️","🤦🏽‍♀️","🤦🏾‍♀️","🤦🏿‍♀️","🤷‍♂️","🤷🏻‍♂️","🤷🏼‍♂️","🤷🏽‍♂️","🤷🏾‍♂️","🤷🏿‍♂️","🤷‍♀️","🤷🏻‍♀️","🤷🏼‍♀️","🤷🏽‍♀️","🤷🏾‍♀️","🤷🏿‍♀️","🧑‍⚕️","🧑🏻‍⚕️","🧑🏼‍⚕️","🧑🏽‍⚕️","🧑🏾‍⚕️","🧑🏿‍⚕️","👨‍⚕️","👨🏻‍⚕️","👨🏼‍⚕️","👨🏽‍⚕️","👨🏾‍⚕️","👨🏿‍⚕️","👩‍⚕️","👩🏻‍⚕️","👩🏼‍⚕️","👩🏽‍⚕️","👩🏾‍⚕️","👩🏿‍⚕️","🧑‍🎓","🧑🏻‍🎓","🧑🏼‍🎓","🧑🏽‍🎓","🧑🏾‍🎓","🧑🏿‍🎓","👨‍🎓","👨🏻‍🎓","👨🏼‍🎓","👨🏽‍🎓","👨🏾‍🎓","👨🏿‍🎓","👩‍🎓","👩🏻‍🎓","👩🏼‍🎓","👩🏽‍🎓","👩🏾‍🎓","👩🏿‍🎓","🧑‍🏫","🧑🏻‍🏫","🧑🏼‍🏫","🧑🏽‍🏫","🧑🏾‍🏫","🧑🏿‍🏫","👨‍🏫","👨🏻‍🏫","👨🏼‍🏫","👨🏽‍🏫","👨🏾‍🏫","👨🏿‍🏫","👩‍🏫","👩🏻‍🏫","👩🏼‍🏫","👩🏽‍🏫","👩🏾‍🏫","👩🏿‍🏫","🧑‍⚖️","🧑🏻‍⚖️","🧑🏼‍⚖️","🧑🏽‍⚖️","🧑🏾‍⚖️","🧑🏿‍⚖️","👨‍⚖️","👨🏻‍⚖️","👨🏼‍⚖️","👨🏽‍⚖️","👨🏾‍⚖️","👨🏿‍⚖️","👩‍⚖️","👩🏻‍⚖️","👩🏼‍⚖️","👩🏽‍⚖️","👩🏾‍⚖️","👩🏿‍⚖️","🧑‍🌾","🧑🏻‍🌾","🧑🏼‍🌾","🧑🏽‍🌾","🧑🏾‍🌾","🧑🏿‍🌾","👨‍🌾","👨🏻‍🌾","👨🏼‍🌾","👨🏽‍🌾","👨🏾‍🌾","👨🏿‍🌾","👩‍🌾","👩🏻‍🌾","👩🏼‍🌾","👩🏽‍🌾","👩🏾‍🌾","👩🏿‍🌾","🧑‍🍳","🧑🏻‍🍳","🧑🏼‍🍳","🧑🏽‍🍳","🧑🏾‍🍳","🧑🏿‍🍳","👨‍🍳","👨🏻‍🍳","👨🏼‍🍳","👨🏽‍🍳","👨🏾‍🍳","👨🏿‍🍳","👩‍🍳","👩🏻‍🍳","👩🏼‍🍳","👩🏽‍🍳","👩🏾‍🍳","👩🏿‍🍳","🧑‍🔧","🧑🏻‍🔧","🧑🏼‍🔧","🧑🏽‍🔧","🧑🏾‍🔧","🧑🏿‍🔧","👨‍🔧","👨🏻‍🔧","👨🏼‍🔧","👨🏽‍🔧","👨🏾‍🔧","👨🏿‍🔧","👩‍🔧","👩🏻‍🔧","👩🏼‍🔧","👩🏽‍🔧","👩🏾‍🔧","👩🏿‍🔧","🧑‍🏭","🧑🏻‍🏭","🧑🏼‍🏭","🧑🏽‍🏭","🧑🏾‍🏭","🧑🏿‍🏭","👨‍🏭","👨🏻‍🏭","👨🏼‍🏭","👨🏽‍🏭","👨🏾‍🏭","👨🏿‍🏭","👩‍🏭","👩🏻‍🏭","👩🏼‍🏭","👩🏽‍🏭","👩🏾‍🏭","👩🏿‍🏭","🧑‍💼","🧑🏻‍💼","🧑🏼‍💼","🧑🏽‍💼","🧑🏾‍💼","🧑🏿‍💼","👨‍💼","👨🏻‍💼","👨🏼‍💼","👨🏽‍💼","👨🏾‍💼","👨🏿‍💼","👩‍💼","👩🏻‍💼","👩🏼‍💼","👩🏽‍💼","👩🏾‍💼","👩🏿‍💼","🧑‍🔬","🧑🏻‍🔬","🧑🏼‍🔬","🧑🏽‍🔬","🧑🏾‍🔬","🧑🏿‍🔬","👨‍🔬","👨🏻‍🔬","👨🏼‍🔬","👨🏽‍🔬","👨🏾‍🔬","👨🏿‍🔬","👩‍🔬","👩🏻‍🔬","👩🏼‍🔬","👩🏽‍🔬","👩🏾‍🔬","👩🏿‍🔬","🧑‍💻","🧑🏻‍💻","🧑🏼‍💻","🧑🏽‍💻","🧑🏾‍💻","🧑🏿‍💻","👨‍💻","👨🏻‍💻","👨🏼‍💻","👨🏽‍💻","👨🏾‍💻","👨🏿‍💻","👩‍💻","👩🏻‍💻","👩🏼‍💻","👩🏽‍💻","👩🏾‍💻","👩🏿‍💻","🧑‍🎤","🧑🏻‍🎤","🧑🏼‍🎤","🧑🏽‍🎤","🧑🏾‍🎤","🧑🏿‍🎤","👨‍🎤","👨🏻‍🎤","👨🏼‍🎤","👨🏽‍🎤","👨🏾‍🎤","👨🏿‍🎤","👩‍🎤","👩🏻‍🎤","👩🏼‍🎤","👩🏽‍🎤","👩🏾‍🎤","👩🏿‍🎤","🧑‍🎨","🧑🏻‍🎨","🧑🏼‍🎨","🧑🏽‍🎨","🧑🏾‍🎨","🧑🏿‍🎨","👨‍🎨","👨🏻‍🎨","👨🏼‍🎨","👨🏽‍🎨","👨🏾‍🎨","👨🏿‍🎨","👩‍🎨","👩🏻‍🎨","👩🏼‍🎨","👩🏽‍🎨","👩🏾‍🎨","👩🏿‍🎨","🧑‍✈️","🧑🏻‍✈️","🧑🏼‍✈️","🧑🏽‍✈️","🧑🏾‍✈️","🧑🏿‍✈️","👨‍✈️","👨🏻‍✈️","👨🏼‍✈️","👨🏽‍✈️","👨🏾‍✈️","👨🏿‍✈️","👩‍✈️","👩🏻‍✈️","👩🏼‍✈️","👩🏽‍✈️","👩🏾‍✈️","👩🏿‍✈️","🧑‍🚀","🧑🏻‍🚀","🧑🏼‍🚀","🧑🏽‍🚀","🧑🏾‍🚀","🧑🏿‍🚀","👨‍🚀","👨🏻‍🚀","👨🏼‍🚀","👨🏽‍🚀","👨🏾‍🚀","👨🏿‍🚀","👩‍🚀","👩🏻‍🚀","👩🏼‍🚀","👩🏽‍🚀","👩🏾‍🚀","👩🏿‍🚀","🧑‍🚒","🧑🏻‍🚒","🧑🏼‍🚒","🧑🏽‍🚒","🧑🏾‍🚒","🧑🏿‍🚒","👨‍🚒","👨🏻‍🚒","👨🏼‍🚒","👨🏽‍🚒","👨🏾‍🚒","👨🏿‍🚒","👩‍🚒","👩🏻‍🚒","👩🏼‍🚒","👩🏽‍🚒","👩🏾‍🚒","👩🏿‍🚒","👮‍♂️","👮🏻‍♂️","👮🏼‍♂️","👮🏽‍♂️","👮🏾‍♂️","👮🏿‍♂️","👮‍♀️","👮🏻‍♀️","👮🏼‍♀️","👮🏽‍♀️","👮🏾‍♀️","👮🏿‍♀️","🕵️‍♂️","🕵🏻‍♂️","🕵🏼‍♂️","🕵🏽‍♂️","🕵🏾‍♂️","🕵🏿‍♂️","🕵️‍♀️","🕵🏻‍♀️","🕵🏼‍♀️","🕵🏽‍♀️","🕵🏾‍♀️","🕵🏿‍♀️","💂‍♂️","💂🏻‍♂️","💂🏼‍♂️","💂🏽‍♂️","💂🏾‍♂️","💂🏿‍♂️","💂‍♀️","💂🏻‍♀️","💂🏼‍♀️","💂🏽‍♀️","💂🏾‍♀️","💂🏿‍♀️","👷‍♂️","👷🏻‍♂️","👷🏼‍♂️","👷🏽‍♂️","👷🏾‍♂️","👷🏿‍♂️","👷‍♀️","👷🏻‍♀️","👷🏼‍♀️","👷🏽‍♀️","👷🏾‍♀️","👷🏿‍♀️","👳‍♂️","👳🏻‍♂️","👳🏼‍♂️","👳🏽‍♂️","👳🏾‍♂️","👳🏿‍♂️","👳‍♀️","👳🏻‍♀️","👳🏼‍♀️","👳🏽‍♀️","👳🏾‍♀️","👳🏿‍♀️","🤵‍♂️","🤵🏻‍♂️","🤵🏼‍♂️","🤵🏽‍♂️","🤵🏾‍♂️","🤵🏿‍♂️","🤵‍♀️","🤵🏻‍♀️","🤵🏼‍♀️","🤵🏽‍♀️","🤵🏾‍♀️","🤵🏿‍♀️","👰‍♂️","👰🏻‍♂️","👰🏼‍♂️","👰🏽‍♂️","👰🏾‍♂️","👰🏿‍♂️","👰‍♀️","👰🏻‍♀️","👰🏼‍♀️","👰🏽‍♀️","👰🏾‍♀️","👰🏿‍♀️","👩‍🍼","👩🏻‍🍼","👩🏼‍🍼","👩🏽‍🍼","👩🏾‍🍼","👩🏿‍🍼","👨‍🍼","👨🏻‍🍼","👨🏼‍🍼","👨🏽‍🍼","👨🏾‍🍼","👨🏿‍🍼","🧑‍🍼","🧑🏻‍🍼","🧑🏼‍🍼","🧑🏽‍🍼","🧑🏾‍🍼","🧑🏿‍🍼","🧑‍🎄","🧑🏻‍🎄","🧑🏼‍🎄","🧑🏽‍🎄","🧑🏾‍🎄","🧑🏿‍🎄","🦸‍♂️","🦸🏻‍♂️","🦸🏼‍♂️","🦸🏽‍♂️","🦸🏾‍♂️","🦸🏿‍♂️","🦸‍♀️","🦸🏻‍♀️","🦸🏼‍♀️","🦸🏽‍♀️","🦸🏾‍♀️","🦸🏿‍♀️","🦹‍♂️","🦹🏻‍♂️","🦹🏼‍♂️","🦹🏽‍♂️","🦹🏾‍♂️","🦹🏿‍♂️","🦹‍♀️","🦹🏻‍♀️","🦹🏼‍♀️","🦹🏽‍♀️","🦹🏾‍♀️","🦹🏿‍♀️","🧙‍♂️","🧙🏻‍♂️","🧙🏼‍♂️","🧙🏽‍♂️","🧙🏾‍♂️","🧙🏿‍♂️","🧙‍♀️","🧙🏻‍♀️","🧙🏼‍♀️","🧙🏽‍♀️","🧙🏾‍♀️","🧙🏿‍♀️","🧚‍♂️","🧚🏻‍♂️","🧚🏼‍♂️","🧚🏽‍♂️","🧚🏾‍♂️","🧚🏿‍♂️","🧚‍♀️","🧚🏻‍♀️","🧚🏼‍♀️","🧚🏽‍♀️","🧚🏾‍♀️","🧚🏿‍♀️","🧛‍♂️","🧛🏻‍♂️","🧛🏼‍♂️","🧛🏽‍♂️","🧛🏾‍♂️","🧛🏿‍♂️","🧛‍♀️","🧛🏻‍♀️","🧛🏼‍♀️","🧛🏽‍♀️","🧛🏾‍♀️","🧛🏿‍♀️","🧜‍♂️","🧜🏻‍♂️","🧜🏼‍♂️","🧜🏽‍♂️","🧜🏾‍♂️","🧜🏿‍♂️","🧜‍♀️","🧜🏻‍♀️","🧜🏼‍♀️","🧜🏽‍♀️","🧜🏾‍♀️","🧜🏿‍♀️","🧝‍♂️","🧝🏻‍♂️","🧝🏼‍♂️","🧝🏽‍♂️","🧝🏾‍♂️","🧝🏿‍♂️","🧝‍♀️","🧝🏻‍♀️","🧝🏼‍♀️","🧝🏽‍♀️","🧝🏾‍♀️","🧝🏿‍♀️","🧞‍♂️","🧞‍♀️","🧟‍♂️","🧟‍♀️","💆‍♂️","💆🏻‍♂️","💆🏼‍♂️","💆🏽‍♂️","💆🏾‍♂️","💆🏿‍♂️","💆‍♀️","💆🏻‍♀️","💆🏼‍♀️","💆🏽‍♀️","💆🏾‍♀️","💆🏿‍♀️","💇‍♂️","💇🏻‍♂️","💇🏼‍♂️","💇🏽‍♂️","💇🏾‍♂️","💇🏿‍♂️","💇‍♀️","💇🏻‍♀️","💇🏼‍♀️","💇🏽‍♀️","💇🏾‍♀️","💇🏿‍♀️","🚶‍♂️","🚶🏻‍♂️","🚶🏼‍♂️","🚶🏽‍♂️","🚶🏾‍♂️","🚶🏿‍♂️","🚶‍♀️","🚶🏻‍♀️","🚶🏼‍♀️","🚶🏽‍♀️","🚶🏾‍♀️","🚶🏿‍♀️","🧍‍♂️","🧍🏻‍♂️","🧍🏼‍♂️","🧍🏽‍♂️","🧍🏾‍♂️","🧍🏿‍♂️","🧍‍♀️","🧍🏻‍♀️","🧍🏼‍♀️","🧍🏽‍♀️","🧍🏾‍♀️","🧍🏿‍♀️","🧎‍♂️","🧎🏻‍♂️","🧎🏼‍♂️","🧎🏽‍♂️","🧎🏾‍♂️","🧎🏿‍♂️","🧎‍♀️","🧎🏻‍♀️","🧎🏼‍♀️","🧎🏽‍♀️","🧎🏾‍♀️","🧎🏿‍♀️","🧑‍🦯","🧑🏻‍🦯","🧑🏼‍🦯","🧑🏽‍🦯","🧑🏾‍🦯","🧑🏿‍🦯","👨‍🦯","👨🏻‍🦯","👨🏼‍🦯","👨🏽‍🦯","👨🏾‍🦯","👨🏿‍🦯","👩‍🦯","👩🏻‍🦯","👩🏼‍🦯","👩🏽‍🦯","👩🏾‍🦯","👩🏿‍🦯","🧑‍🦼","🧑🏻‍🦼","🧑🏼‍🦼","🧑🏽‍🦼","🧑🏾‍🦼","🧑🏿‍🦼","👨‍🦼","👨🏻‍🦼","👨🏼‍🦼","👨🏽‍🦼","👨🏾‍🦼","👨🏿‍🦼","👩‍🦼","👩🏻‍🦼","👩🏼‍🦼","👩🏽‍🦼","👩🏾‍🦼","👩🏿‍🦼","🧑‍🦽","🧑🏻‍🦽","🧑🏼‍🦽","🧑🏽‍🦽","🧑🏾‍🦽","🧑🏿‍🦽","👨‍🦽","👨🏻‍🦽","👨🏼‍🦽","👨🏽‍🦽","👨🏾‍🦽","👨🏿‍🦽","👩‍🦽","👩🏻‍🦽","👩🏼‍🦽","👩🏽‍🦽","👩🏾‍🦽","👩🏿‍🦽","🏃‍♂️","🏃🏻‍♂️","🏃🏼‍♂️","🏃🏽‍♂️","🏃🏾‍♂️","🏃🏿‍♂️","🏃‍♀️","🏃🏻‍♀️","🏃🏼‍♀️","🏃🏽‍♀️","🏃🏾‍♀️","🏃🏿‍♀️","👯‍♂️","👯‍♀️","🧖‍♂️","🧖🏻‍♂️","🧖🏼‍♂️","🧖🏽‍♂️","🧖🏾‍♂️","🧖🏿‍♂️","🧖‍♀️","🧖🏻‍♀️","🧖🏼‍♀️","🧖🏽‍♀️","🧖🏾‍♀️","🧖🏿‍♀️","🧗‍♂️","🧗🏻‍♂️","🧗🏼‍♂️","🧗🏽‍♂️","🧗🏾‍♂️","🧗🏿‍♂️","🧗‍♀️","🧗🏻‍♀️","🧗🏼‍♀️","🧗🏽‍♀️","🧗🏾‍♀️","🧗🏿‍♀️","🏌️‍♂️","🏌🏻‍♂️","🏌🏼‍♂️","🏌🏽‍♂️","🏌🏾‍♂️","🏌🏿‍♂️","🏌️‍♀️","🏌🏻‍♀️","🏌🏼‍♀️","🏌🏽‍♀️","🏌🏾‍♀️","🏌🏿‍♀️","🏄‍♂️","🏄🏻‍♂️","🏄🏼‍♂️","🏄🏽‍♂️","🏄🏾‍♂️","🏄🏿‍♂️","🏄‍♀️","🏄🏻‍♀️","🏄🏼‍♀️","🏄🏽‍♀️","🏄🏾‍♀️","🏄🏿‍♀️","🚣‍♂️","🚣🏻‍♂️","🚣🏼‍♂️","🚣🏽‍♂️","🚣🏾‍♂️","🚣🏿‍♂️","🚣‍♀️","🚣🏻‍♀️","🚣🏼‍♀️","🚣🏽‍♀️","🚣🏾‍♀️","🚣🏿‍♀️","🏊‍♂️","🏊🏻‍♂️","🏊🏼‍♂️","🏊🏽‍♂️","🏊🏾‍♂️","🏊🏿‍♂️","🏊‍♀️","🏊🏻‍♀️","🏊🏼‍♀️","🏊🏽‍♀️","🏊🏾‍♀️","🏊🏿‍♀️","⛹️‍♂️","⛹🏻‍♂️","⛹🏼‍♂️","⛹🏽‍♂️","⛹🏾‍♂️","⛹🏿‍♂️","⛹️‍♀️","⛹🏻‍♀️","⛹🏼‍♀️","⛹🏽‍♀️","⛹🏾‍♀️","⛹🏿‍♀️","🏋️‍♂️","🏋🏻‍♂️","🏋🏼‍♂️","🏋🏽‍♂️","🏋🏾‍♂️","🏋🏿‍♂️","🏋️‍♀️","🏋🏻‍♀️","🏋🏼‍♀️","🏋🏽‍♀️","🏋🏾‍♀️","🏋🏿‍♀️","🚴‍♂️","🚴🏻‍♂️","🚴🏼‍♂️","🚴🏽‍♂️","🚴🏾‍♂️","🚴🏿‍♂️","🚴‍♀️","🚴🏻‍♀️","🚴🏼‍♀️","🚴🏽‍♀️","🚴🏾‍♀️","🚴🏿‍♀️","🚵‍♂️","🚵🏻‍♂️","🚵🏼‍♂️","🚵🏽‍♂️","🚵🏾‍♂️","🚵🏿‍♂️","🚵‍♀️","🚵🏻‍♀️","🚵🏼‍♀️","🚵🏽‍♀️","🚵🏾‍♀️","🚵🏿‍♀️","🤸‍♂️","🤸🏻‍♂️","🤸🏼‍♂️","🤸🏽‍♂️","🤸🏾‍♂️","🤸🏿‍♂️","🤸‍♀️","🤸🏻‍♀️","🤸🏼‍♀️","🤸🏽‍♀️","🤸🏾‍♀️","🤸🏿‍♀️","🤼‍♂️","🤼‍♀️","🤽‍♂️","🤽🏻‍♂️","🤽🏼‍♂️","🤽🏽‍♂️","🤽🏾‍♂️","🤽🏿‍♂️","🤽‍♀️","🤽🏻‍♀️","🤽🏼‍♀️","🤽🏽‍♀️","🤽🏾‍♀️","🤽🏿‍♀️","🤾‍♂️","🤾🏻‍♂️","🤾🏼‍♂️","🤾🏽‍♂️","🤾🏾‍♂️","🤾🏿‍♂️","🤾‍♀️","🤾🏻‍♀️","🤾🏼‍♀️","🤾🏽‍♀️","🤾🏾‍♀️","🤾🏿‍♀️","🤹‍♂️","🤹🏻‍♂️","🤹🏼‍♂️","🤹🏽‍♂️","🤹🏾‍♂️","🤹🏿‍♂️","🤹‍♀️","🤹🏻‍♀️","🤹🏼‍♀️","🤹🏽‍♀️","🤹🏾‍♀️","🤹🏿‍♀️","🧘‍♂️","🧘🏻‍♂️","🧘🏼‍♂️","🧘🏽‍♂️","🧘🏾‍♂️","🧘🏿‍♂️","🧘‍♀️","🧘🏻‍♀️","🧘🏼‍♀️","🧘🏽‍♀️","🧘🏾‍♀️","🧘🏿‍♀️","🧑‍🤝‍🧑","🧑🏻‍🤝‍🧑🏻","🧑🏻‍🤝‍🧑🏼","🧑🏻‍🤝‍🧑🏽","🧑🏻‍🤝‍🧑🏾","🧑🏻‍🤝‍🧑🏿","🧑🏼‍🤝‍🧑🏻","🧑🏼‍🤝‍🧑🏼","🧑🏼‍🤝‍🧑🏽","🧑🏼‍🤝‍🧑🏾","🧑🏼‍🤝‍🧑🏿","🧑🏽‍🤝‍🧑🏻","🧑🏽‍🤝‍🧑🏼","🧑🏽‍🤝‍🧑🏽","🧑🏽‍🤝‍🧑🏾","🧑🏽‍🤝‍🧑🏿","🧑🏾‍🤝‍🧑🏻","🧑🏾‍🤝‍🧑🏼","🧑🏾‍🤝‍🧑🏽","🧑🏾‍🤝‍🧑🏾","🧑🏾‍🤝‍🧑🏿","🧑🏿‍🤝‍🧑🏻","🧑🏿‍🤝‍🧑🏼","🧑🏿‍🤝‍🧑🏽","🧑🏿‍🤝‍🧑🏾","🧑🏿‍🤝‍🧑🏿","👩🏻‍🤝‍👩🏼","👩🏻‍🤝‍👩🏽","👩🏻‍🤝‍👩🏾","👩🏻‍🤝‍👩🏿","👩🏼‍🤝‍👩🏻","👩🏼‍🤝‍👩🏽","👩🏼‍🤝‍👩🏾","👩🏼‍🤝‍👩🏿","👩🏽‍🤝‍👩🏻","👩🏽‍🤝‍👩🏼","👩🏽‍🤝‍👩🏾","👩🏽‍🤝‍👩🏿","👩🏾‍🤝‍👩🏻","👩🏾‍🤝‍👩🏼","👩🏾‍🤝‍👩🏽","👩🏾‍🤝‍👩🏿","👩🏿‍🤝‍👩🏻","👩🏿‍🤝‍👩🏼","👩🏿‍🤝‍👩🏽","👩🏿‍🤝‍👩🏾","👩🏻‍🤝‍👨🏼","👩🏻‍🤝‍👨🏽","👩🏻‍🤝‍👨🏾","👩🏻‍🤝‍👨🏿","👩🏼‍🤝‍👨🏻","👩🏼‍🤝‍👨🏽","👩🏼‍🤝‍👨🏾","👩🏼‍🤝‍👨🏿","👩🏽‍🤝‍👨🏻","👩🏽‍🤝‍👨🏼","👩🏽‍🤝‍👨🏾","👩🏽‍🤝‍👨🏿","👩🏾‍🤝‍👨🏻","👩🏾‍🤝‍👨🏼","👩🏾‍🤝‍👨🏽","👩🏾‍🤝‍👨🏿","👩🏿‍🤝‍👨🏻","👩🏿‍🤝‍👨🏼","👩🏿‍🤝‍👨🏽","👩🏿‍🤝‍👨🏾","👨🏻‍🤝‍👨🏼","👨🏻‍🤝‍👨🏽","👨🏻‍🤝‍👨🏾","👨🏻‍🤝‍👨🏿","👨🏼‍🤝‍👨🏻","👨🏼‍🤝‍👨🏽","👨🏼‍🤝‍👨🏾","👨🏼‍🤝‍👨🏿","👨🏽‍🤝‍👨🏻","👨🏽‍🤝‍👨🏼","👨🏽‍🤝‍👨🏾","👨🏽‍🤝‍👨🏿","👨🏾‍🤝‍👨🏻","👨🏾‍🤝‍👨🏼","👨🏾‍🤝‍👨🏽","👨🏾‍🤝‍👨🏿","👨🏿‍🤝‍👨🏻","👨🏿‍🤝‍👨🏼","👨🏿‍🤝‍👨🏽","👨🏿‍🤝‍👨🏾","💏","🧑🏻‍❤️‍💋‍🧑🏼","🧑🏻‍❤️‍💋‍🧑🏽","🧑🏻‍❤️‍💋‍🧑🏾","🧑🏻‍❤️‍💋‍🧑🏿","🧑🏼‍❤️‍💋‍🧑🏻","🧑🏼‍❤️‍💋‍🧑🏽","🧑🏼‍❤️‍💋‍🧑🏾","🧑🏼‍❤️‍💋‍🧑🏿","🧑🏽‍❤️‍💋‍🧑🏻","🧑🏽‍❤️‍💋‍🧑🏼","🧑🏽‍❤️‍💋‍🧑🏾","🧑🏽‍❤️‍💋‍🧑🏿","🧑🏾‍❤️‍💋‍🧑🏻","🧑🏾‍❤️‍💋‍🧑🏼","🧑🏾‍❤️‍💋‍🧑🏽","🧑🏾‍❤️‍💋‍🧑🏿","🧑🏿‍❤️‍💋‍🧑🏻","🧑🏿‍❤️‍💋‍🧑🏼","🧑🏿‍❤️‍💋‍🧑🏽","🧑🏿‍❤️‍💋‍🧑🏾","👩‍❤️‍💋‍👨","👩🏻‍❤️‍💋‍👨🏻","👩🏻‍❤️‍💋‍👨🏼","👩🏻‍❤️‍💋‍👨🏽","👩🏻‍❤️‍💋‍👨🏾","👩🏻‍❤️‍💋‍👨🏿","👩🏼‍❤️‍💋‍👨🏻","👩🏼‍❤️‍💋‍👨🏼","👩🏼‍❤️‍💋‍👨🏽","👩🏼‍❤️‍💋‍👨🏾","👩🏼‍❤️‍💋‍👨🏿","👩🏽‍❤️‍💋‍👨🏻","👩🏽‍❤️‍💋‍👨🏼","👩🏽‍❤️‍💋‍👨🏽","👩🏽‍❤️‍💋‍👨🏾","👩🏽‍❤️‍💋‍👨🏿","👩🏾‍❤️‍💋‍👨🏻","👩🏾‍❤️‍💋‍👨🏼","👩🏾‍❤️‍💋‍👨🏽","👩🏾‍❤️‍💋‍👨🏾","👩🏾‍❤️‍💋‍👨🏿","👩🏿‍❤️‍💋‍👨🏻","👩🏿‍❤️‍💋‍👨🏼","👩🏿‍❤️‍💋‍👨🏽","👩🏿‍❤️‍💋‍👨🏾","👩🏿‍❤️‍💋‍👨🏿","👨‍❤️‍💋‍👨","👨🏻‍❤️‍💋‍👨🏻","👨🏻‍❤️‍💋‍👨🏼","👨🏻‍❤️‍💋‍👨🏽","👨🏻‍❤️‍💋‍👨🏾","👨🏻‍❤️‍💋‍👨🏿","👨🏼‍❤️‍💋‍👨🏻","👨🏼‍❤️‍💋‍👨🏼","👨🏼‍❤️‍💋‍👨🏽","👨🏼‍❤️‍💋‍👨🏾","👨🏼‍❤️‍💋‍👨🏿","👨🏽‍❤️‍💋‍👨🏻","👨🏽‍❤️‍💋‍👨🏼","👨🏽‍❤️‍💋‍👨🏽","👨🏽‍❤️‍💋‍👨🏾","👨🏽‍❤️‍💋‍👨🏿","👨🏾‍❤️‍💋‍👨🏻","👨🏾‍❤️‍💋‍👨🏼","👨🏾‍❤️‍💋‍👨🏽","👨🏾‍❤️‍💋‍👨🏾","👨🏾‍❤️‍💋‍👨🏿","👨🏿‍❤️‍💋‍👨🏻","👨🏿‍❤️‍💋‍👨🏼","👨🏿‍❤️‍💋‍👨🏽","👨🏿‍❤️‍💋‍👨🏾","👨🏿‍❤️‍💋‍👨🏿","👩‍❤️‍💋‍👩","👩🏻‍❤️‍💋‍👩🏻","👩🏻‍❤️‍💋‍👩🏼","👩🏻‍❤️‍💋‍👩🏽","👩🏻‍❤️‍💋‍👩🏾","👩🏻‍❤️‍💋‍👩🏿","👩🏼‍❤️‍💋‍👩🏻","👩🏼‍❤️‍💋‍👩🏼","👩🏼‍❤️‍💋‍👩🏽","👩🏼‍❤️‍💋‍👩🏾","👩🏼‍❤️‍💋‍👩🏿","👩🏽‍❤️‍💋‍👩🏻","👩🏽‍❤️‍💋‍👩🏼","👩🏽‍❤️‍💋‍👩🏽","👩🏽‍❤️‍💋‍👩🏾","👩🏽‍❤️‍💋‍👩🏿","👩🏾‍❤️‍💋‍👩🏻","👩🏾‍❤️‍💋‍👩🏼","👩🏾‍❤️‍💋‍👩🏽","👩🏾‍❤️‍💋‍👩🏾","👩🏾‍❤️‍💋‍👩🏿","👩🏿‍❤️‍💋‍👩🏻","👩🏿‍❤️‍💋‍👩🏼","👩🏿‍❤️‍💋‍👩🏽","👩🏿‍❤️‍💋‍👩🏾","👩🏿‍❤️‍💋‍👩🏿","💑","🧑🏻‍❤️‍🧑🏼","🧑🏻‍❤️‍🧑🏽","🧑🏻‍❤️‍🧑🏾","🧑🏻‍❤️‍🧑🏿","🧑🏼‍❤️‍🧑🏻","🧑🏼‍❤️‍🧑🏽","🧑🏼‍❤️‍🧑🏾","🧑🏼‍❤️‍🧑🏿","🧑🏽‍❤️‍🧑🏻","🧑🏽‍❤️‍🧑🏼","🧑🏽‍❤️‍🧑🏾","🧑🏽‍❤️‍🧑🏿","🧑🏾‍❤️‍🧑🏻","🧑🏾‍❤️‍🧑🏼","🧑🏾‍❤️‍🧑🏽","🧑🏾‍❤️‍🧑🏿","🧑🏿‍❤️‍🧑🏻","🧑🏿‍❤️‍🧑🏼","🧑🏿‍❤️‍🧑🏽","🧑🏿‍❤️‍🧑🏾","👩‍❤️‍👨","👩🏻‍❤️‍👨🏻","👩🏻‍❤️‍👨🏼","👩🏻‍❤️‍👨🏽","👩🏻‍❤️‍👨🏾","👩🏻‍❤️‍👨🏿","👩🏼‍❤️‍👨🏻","👩🏼‍❤️‍👨🏼","👩🏼‍❤️‍👨🏽","👩🏼‍❤️‍👨🏾","👩🏼‍❤️‍👨🏿","👩🏽‍❤️‍👨🏻","👩🏽‍❤️‍👨🏼","👩🏽‍❤️‍👨🏽","👩🏽‍❤️‍👨🏾","👩🏽‍❤️‍👨🏿","👩🏾‍❤️‍👨🏻","👩🏾‍❤️‍👨🏼","👩🏾‍❤️‍👨🏽","👩🏾‍❤️‍👨🏾","👩🏾‍❤️‍👨🏿","👩🏿‍❤️‍👨🏻","👩🏿‍❤️‍👨🏼","👩🏿‍❤️‍👨🏽","👩🏿‍❤️‍👨🏾","👩🏿‍❤️‍👨🏿","👨‍❤️‍👨","👨🏻‍❤️‍👨🏻","👨🏻‍❤️‍👨🏼","👨🏻‍❤️‍👨🏽","👨🏻‍❤️‍👨🏾","👨🏻‍❤️‍👨🏿","👨🏼‍❤️‍👨🏻","👨🏼‍❤️‍👨🏼","👨🏼‍❤️‍👨🏽","👨🏼‍❤️‍👨🏾","👨🏼‍❤️‍👨🏿","👨🏽‍❤️‍👨🏻","👨🏽‍❤️‍👨🏼","👨🏽‍❤️‍👨🏽","👨🏽‍❤️‍👨🏾","👨🏽‍❤️‍👨🏿","👨🏾‍❤️‍👨🏻","👨🏾‍❤️‍👨🏼","👨🏾‍❤️‍👨🏽","👨🏾‍❤️‍👨🏾","👨🏾‍❤️‍👨🏿","👨🏿‍❤️‍👨🏻","👨🏿‍❤️‍👨🏼","👨🏿‍❤️‍👨🏽","👨🏿‍❤️‍👨🏾","👨🏿‍❤️‍👨🏿","👩‍❤️‍👩","👩🏻‍❤️‍👩🏻","👩🏻‍❤️‍👩🏼","👩🏻‍❤️‍👩🏽","👩🏻‍❤️‍👩🏾","👩🏻‍❤️‍👩🏿","👩🏼‍❤️‍👩🏻","👩🏼‍❤️‍👩🏼","👩🏼‍❤️‍👩🏽","👩🏼‍❤️‍👩🏾","👩🏼‍❤️‍👩🏿","👩🏽‍❤️‍👩🏻","👩🏽‍❤️‍👩🏼","👩🏽‍❤️‍👩🏽","👩🏽‍❤️‍👩🏾","👩🏽‍❤️‍👩🏿","👩🏾‍❤️‍👩🏻","👩🏾‍❤️‍👩🏼","👩🏾‍❤️‍👩🏽","👩🏾‍❤️‍👩🏾","👩🏾‍❤️‍👩🏿","👩🏿‍❤️‍👩🏻","👩🏿‍❤️‍👩🏼","👩🏿‍❤️‍👩🏽","👩🏿‍❤️‍👩🏾","👩🏿‍❤️‍👩🏿","👪","👨‍👩‍👦","👨‍👩‍👧","👨‍👩‍👧‍👦","👨‍👩‍👦‍👦","👨‍👩‍👧‍👧","👨‍👨‍👦","👨‍👨‍👧","👨‍👨‍👧‍👦","👨‍👨‍👦‍👦","👨‍👨‍👧‍👧","👩‍👩‍👦","👩‍👩‍👧","👩‍👩‍👧‍👦","👩‍👩‍👦‍👦","👩‍👩‍👧‍👧","👨‍👦","👨‍👦‍👦","👨‍👧","👨‍👧‍👦","👨‍👧‍👧","👩‍👦","👩‍👦‍👦","👩‍👧","👩‍👧‍👦","👩‍👧‍👧","🐕‍🦺","🐈‍⬛","🐻‍❄️","🏳️‍🌈","🏳️‍⚧️","🏴‍☠️"],"expected":[{"id":"0","event":null,"data":"Blåbæ\nrsyltetøy må da være mulig å oppdrive? 😶‍🌫️ 😮‍💨 😵‍💫 ❤️‍🔥 ❤️‍🩹 👁️‍🗨️ 🫱🏻‍🫲🏼 🫱🏻‍🫲🏽 🫱🏻‍🫲🏾 🫱🏻‍🫲🏿 🫱🏼‍🫲🏻 🫱🏼‍🫲🏽 🫱🏼‍🫲🏾 🫱🏼‍🫲🏿 🫱🏽‍🫲🏻 🫱🏽‍🫲🏼 🫱🏽‍🫲🏾 🫱🏽‍🫲🏿 🫱🏾‍🫲🏻 🫱🏾‍🫲🏼 🫱🏾‍🫲🏽 🫱🏾‍🫲🏿 🫱🏿‍🫲🏻 🫱🏿‍🫲🏼 🫱🏿‍🫲🏽 🫱🏿‍🫲🏾 🧔‍♂️ 🧔🏻‍♂️ 🧔🏼‍♂️ 🧔🏽‍♂️ 🧔🏾‍♂️ 🧔🏿‍♂️ 🧔‍♀️ 🧔🏻‍♀️ 🧔🏼‍♀️ 🧔🏽‍♀️ 🧔🏾‍♀️ 🧔🏿‍♀️ 👨‍🦰 👨🏻‍🦰 👨🏼‍🦰 👨🏽‍🦰 👨🏾‍🦰 👨🏿‍🦰 👨‍🦱 👨🏻‍🦱 👨🏼‍🦱 👨🏽‍🦱 👨🏾‍🦱 👨🏿‍🦱 👨‍🦳 👨🏻‍🦳 👨🏼‍🦳 👨🏽‍🦳 👨🏾‍🦳 👨🏿‍🦳 👨‍🦲 👨🏻‍🦲 👨🏼‍🦲 👨🏽‍🦲 👨🏾‍🦲 👨🏿‍🦲 👩‍🦰 👩🏻‍🦰 👩🏼‍🦰 👩🏽‍🦰 👩🏾‍🦰 👩🏿‍🦰 🧑‍🦰 🧑🏻‍🦰 🧑🏼‍🦰 🧑🏽‍🦰 🧑🏾‍🦰 🧑🏿‍🦰 👩‍🦱 👩🏻‍🦱 👩🏼‍🦱 👩🏽‍🦱 👩🏾‍🦱 👩🏿‍🦱 🧑‍🦱 🧑🏻‍🦱 🧑🏼‍🦱 🧑🏽‍🦱 🧑🏾‍🦱 🧑🏿‍🦱 👩‍🦳 👩🏻‍🦳 👩🏼‍🦳 👩🏽‍🦳 👩🏾‍🦳 👩🏿‍🦳 🧑‍🦳 🧑🏻‍🦳 🧑🏼‍🦳 🧑🏽‍🦳 🧑🏾‍🦳"},{"id":"1","event":null,"data":"በእርግጥ አንድ ሰው ብሉቤሪ ጃምን መግዛት መቻል አለበት? 🧑🏿‍🦳 👩‍🦲 👩🏻‍🦲 👩🏼‍🦲 👩🏽‍🦲 👩🏾‍🦲 👩🏿‍🦲 🧑‍🦲 🧑🏻‍🦲 🧑🏼‍🦲 🧑🏽‍🦲 🧑🏾‍🦲 🧑🏿‍🦲 👱‍♀️ 👱🏻‍♀️ 👱🏼‍♀️ 👱🏽‍♀️ 👱🏾‍♀️ 👱🏿‍♀️ 👱‍♂️ 👱🏻‍♂️ 👱🏼‍♂️ 👱🏽‍♂️ 👱🏾‍♂️ 👱🏿‍♂️ 🙍‍♂️ 🙍🏻‍♂️ 🙍🏼‍♂️ 🙍🏽‍♂️ 🙍🏾‍♂️ 🙍🏿‍♂️ 🙍‍♀️ 🙍🏻‍♀️ 🙍🏼‍♀️ 🙍🏽‍♀️ 🙍🏾‍♀️ 🙍🏿‍♀️ 🙎‍♂️ 🙎🏻‍♂️ 🙎🏼‍♂️ 🙎🏽‍♂️ 🙎🏾‍♂️ 🙎🏿‍♂️ 🙎‍♀️ 🙎🏻‍♀️ 🙎🏼‍♀️ 🙎🏽‍♀️ 🙎🏾‍♀️ 🙎🏿‍♀️ 🙅‍♂️ 🙅🏻‍♂️ 🙅🏼‍♂️ 🙅🏽‍♂️ 🙅🏾‍♂️ 🙅🏿‍♂️ 🙅‍♀️ 🙅🏻‍♀️ 🙅🏼‍♀️ 🙅🏽‍♀️ 🙅🏾‍♀️ 🙅🏿‍♀️ 🙆‍♂️ 🙆🏻‍♂️ 🙆🏼‍♂️ 🙆🏽‍♂️ 🙆🏾‍♂️ 🙆🏿‍♂️ 🙆‍♀️ 🙆🏻‍♀️ 🙆🏼‍♀️ 🙆🏽‍♀️ 🙆🏾‍♀️ 🙆🏿‍♀️ 💁‍♂️ 💁🏻‍♂️ 💁🏼‍♂️ 💁🏽‍♂️ 💁🏾‍♂️ 💁🏿‍♂️ 💁‍♀️ 💁🏻‍♀️ 💁🏼‍♀️ 💁🏽‍♀️ 💁🏾‍♀️ 💁🏿‍♀️ 🙋‍♂️ 🙋🏻‍♂️ 🙋🏼‍♂️ 🙋🏽‍♂️ 🙋🏾‍♂️ 🙋🏿‍♂️ 🙋‍♀️ 🙋🏻‍♀️ 🙋🏼‍♀️ 🙋🏽‍♀️ 🙋🏾‍♀️ 🙋🏿‍♀️"},{"id":"2","event":null,"data":"بالتأ\nكيد يجب أن يكون المرء قادرا على شراء مربى التوت؟ 🧏‍♂️ 🧏🏻‍♂️ 🧏🏼‍♂️ 🧏🏽‍♂️ 🧏🏾‍♂️ 🧏🏿‍♂️ 🧏‍♀️ 🧏🏻‍♀️ 🧏🏼‍♀️ 🧏🏽‍♀️ 🧏🏾‍♀️ 🧏🏿‍♀️ 🙇‍♂️ 🙇🏻‍♂️ 🙇🏼‍♂️ 🙇🏽‍♂️ 🙇🏾‍♂️ 🙇🏿‍♂️ 🙇‍♀️ 🙇🏻‍♀️ 🙇🏼‍♀️ 🙇🏽‍♀️ 🙇🏾‍♀️ 🙇🏿‍♀️ 🤦‍♂️ 🤦🏻‍♂️ 🤦🏼‍♂️ 🤦🏽‍♂️ 🤦🏾‍♂️ 🤦🏿‍♂️ 🤦‍♀️ 🤦🏻‍♀️ 🤦🏼‍♀️ 🤦🏽‍♀️ 🤦🏾‍♀️ 🤦🏿‍♀️ 🤷‍♂️ 🤷🏻‍♂️ 🤷🏼‍♂️ 🤷🏽‍♂️ 🤷🏾‍♂️ 🤷🏿‍♂️ 🤷‍♀️ 🤷🏻‍♀️ 🤷🏼‍♀️ 🤷🏽‍♀️ 🤷🏾‍♀️ 🤷🏿‍♀️ 🧑‍⚕️ 🧑🏻‍⚕️ 🧑🏼‍⚕️ 🧑🏽‍⚕️ 🧑🏾‍⚕️ 🧑🏿‍⚕️ 👨‍⚕️ 👨🏻‍⚕️ 👨🏼‍⚕️ 👨🏽‍⚕️ 👨🏾‍⚕️ 👨🏿‍⚕️ 👩‍⚕️ 👩🏻‍⚕️ 👩🏼‍⚕️ 👩🏽‍⚕️ 👩🏾‍⚕️ 👩🏿‍⚕️ 🧑‍🎓 🧑🏻‍🎓 🧑🏼‍🎓 🧑🏽‍🎓 🧑🏾‍🎓 🧑🏿‍🎓 👨‍🎓 👨🏻‍🎓 👨🏼‍🎓 👨🏽‍🎓 👨🏾‍🎓 👨🏿‍🎓 👩‍🎓 👩🏻‍🎓 👩🏼‍🎓 👩🏽‍🎓 👩🏾‍🎓 👩🏿‍🎓 🧑‍🏫 🧑🏻‍🏫 🧑🏼‍🏫 🧑🏽‍🏫 🧑🏾‍🏫 🧑🏿‍🏫 👨‍🏫 👨🏻‍🏫 👨🏼‍🏫 👨🏽‍🏫 👨🏾‍🏫 👨🏿‍🏫 👩‍🏫"},{"id":"3","event":null,"data":"一定能买到蓝莓果酱吗? 👩🏻‍🏫 👩🏼‍🏫 👩🏽‍🏫 👩🏾‍🏫 👩🏿‍🏫 🧑‍⚖️ 🧑🏻‍⚖️ 🧑🏼‍⚖️ 🧑🏽‍⚖️ 🧑🏾‍⚖️ 🧑🏿‍⚖️ 👨‍⚖️ 👨🏻‍⚖️ 👨🏼‍⚖️ 👨🏽‍⚖️ 👨🏾‍⚖️ 👨🏿‍⚖️ 👩‍⚖️ 👩🏻‍⚖️ 👩🏼‍⚖️ 👩🏽‍⚖️ 👩🏾‍⚖️ 👩🏿‍⚖️ 🧑‍🌾 🧑🏻‍🌾 🧑🏼‍🌾 🧑🏽‍🌾 🧑🏾‍🌾 🧑🏿‍🌾 👨‍🌾 👨🏻‍🌾 👨🏼‍🌾 👨🏽‍🌾 👨🏾‍🌾 👨🏿‍🌾 👩‍🌾 👩🏻‍🌾 👩🏼‍🌾 👩🏽‍🌾 👩🏾‍🌾 👩🏿‍🌾 🧑‍🍳 🧑🏻‍🍳 🧑🏼‍🍳 🧑🏽‍🍳 🧑🏾‍🍳 🧑🏿‍🍳 👨‍🍳 👨🏻‍🍳 👨🏼‍🍳 👨🏽‍🍳 👨🏾‍🍳 👨🏿‍🍳 👩‍🍳 👩🏻‍🍳 👩🏼‍🍳 👩🏽‍🍳 👩🏾‍🍳 👩🏿‍🍳 🧑‍🔧 🧑🏻‍🔧 🧑🏼‍🔧 🧑🏽‍🔧 🧑🏾‍🔧 🧑🏿‍🔧 👨‍🔧 👨🏻‍🔧 👨🏼‍🔧 👨🏽‍🔧 👨🏾‍🔧 👨🏿‍🔧 👩‍🔧 👩🏻‍🔧 👩🏼‍🔧 👩🏽‍🔧 👩🏾‍🔧 👩🏿‍🔧 🧑‍🏭 🧑🏻‍🏭 🧑🏼‍🏭 🧑🏽‍🏭 🧑🏾‍🏭 🧑🏿‍🏭 👨‍🏭 👨🏻‍🏭 👨🏼‍🏭 👨🏽‍🏭 👨🏾‍🏭 👨🏿‍🏭 👩‍🏭 👩🏻‍🏭 👩🏼‍🏭 👩🏽‍🏭 👩🏾‍🏭 👩🏿‍🏭 🧑‍💼 🧑🏻‍💼"},{"id":"4","event":null,"data":"ચોક્ક\nસ એક બ્લુબેરી જામ મેળવવા માટે સમર્થ હોવા જ જોઈએ? 🧑🏼‍💼 🧑🏽‍💼 🧑🏾‍💼 🧑🏿‍💼 👨‍💼 👨🏻‍💼 👨🏼‍💼 👨🏽‍💼 👨🏾‍💼 👨🏿‍💼 👩‍💼 👩🏻‍💼 👩🏼‍💼 👩🏽‍💼 👩🏾‍💼 👩🏿‍💼 🧑‍🔬 🧑🏻‍🔬 🧑🏼‍🔬 🧑🏽‍🔬 🧑🏾‍🔬 🧑🏿‍🔬 👨‍🔬 👨🏻‍🔬 👨🏼‍🔬 👨🏽‍🔬 👨🏾‍🔬 👨🏿‍🔬 👩‍🔬 👩🏻‍🔬 👩🏼‍🔬 👩🏽‍🔬 👩🏾‍🔬 👩🏿‍🔬 🧑‍💻 🧑🏻‍💻 🧑🏼‍💻 🧑🏽‍💻 🧑🏾‍💻 🧑🏿‍💻 👨‍💻 👨🏻‍💻 👨🏼‍💻 👨🏽‍💻 👨🏾‍💻 👨🏿‍💻 👩‍💻 👩🏻‍💻 👩🏼‍💻 👩🏽‍💻 👩🏾‍💻 👩🏿‍💻 🧑‍🎤 🧑🏻‍🎤 🧑🏼‍🎤 🧑🏽‍🎤 🧑🏾‍🎤 🧑🏿‍🎤 👨‍🎤 👨🏻‍🎤 👨🏼‍🎤 👨🏽‍🎤 👨🏾‍🎤 👨🏿‍🎤 👩‍🎤 👩🏻‍🎤 👩🏼‍🎤 👩🏽‍🎤 👩🏾‍🎤 👩🏿‍🎤 🧑‍🎨 🧑🏻‍🎨 🧑🏼‍🎨 🧑🏽‍🎨 🧑🏾‍🎨 🧑🏿‍🎨 👨‍🎨 👨🏻‍🎨 👨🏼‍🎨 👨🏽‍🎨 👨🏾‍🎨 👨🏿‍🎨 👩‍🎨 👩🏻‍🎨 👩🏼‍🎨 👩🏽‍🎨 👩🏾‍🎨 👩🏿‍🎨 🧑‍✈️ 🧑🏻‍✈️ 🧑🏼‍✈️ 🧑🏽‍✈️ 🧑🏾‍✈️ 🧑🏿‍✈️ 👨‍✈️ 👨🏻‍✈️ 👨🏼‍✈️"},{"id":"5","event":null,"data":"בטח אפשר להשיג ריבת אוכמניות? 👨🏽‍✈️ 👨🏾‍✈️ 👨🏿‍✈️ 👩‍✈️ 👩🏻‍✈️ 👩🏼‍✈️ 👩🏽‍✈️ 👩🏾‍✈️ 👩🏿‍✈️ 🧑‍🚀 🧑🏻‍🚀 🧑🏼‍🚀 🧑🏽‍🚀 🧑🏾‍🚀 🧑🏿‍🚀 👨‍🚀 👨🏻‍🚀 👨🏼‍🚀 👨🏽‍🚀 👨🏾‍🚀 👨🏿‍🚀 👩‍🚀 👩🏻‍🚀 👩🏼‍🚀 👩🏽‍🚀 👩🏾‍🚀 👩🏿‍🚀 🧑‍🚒 🧑🏻‍🚒 🧑🏼‍🚒 🧑🏽‍🚒 🧑🏾‍🚒 🧑🏿‍🚒 👨‍🚒 👨🏻‍🚒 👨🏼‍🚒 👨🏽‍🚒 👨🏾‍🚒 👨🏿‍🚒 👩‍🚒 👩🏻‍🚒 👩🏼‍🚒 👩🏽‍🚒 👩🏾‍🚒 👩🏿‍🚒 👮‍♂️ 👮🏻‍♂️ 👮🏼‍♂️ 👮🏽‍♂️ 👮🏾‍♂️ 👮🏿‍♂️ 👮‍♀️ 👮🏻‍♀️ 👮🏼‍♀️ 👮🏽‍♀️ 👮🏾‍♀️ 👮🏿‍♀️ 🕵️‍♂️ 🕵🏻‍♂️ 🕵🏼‍♂️ 🕵🏽‍♂️ 🕵🏾‍♂️ 🕵🏿‍♂️ 🕵️‍♀️ 🕵🏻‍♀️ 🕵🏼‍♀️ 🕵🏽‍♀️ 🕵🏾‍♀️ 🕵🏿‍♀️ 💂‍♂️ 💂🏻‍♂️ 💂🏼‍♂️ 💂🏽‍♂️ 💂🏾‍♂️ 💂🏿‍♂️ 💂‍♀️ 💂🏻‍♀️ 💂🏼‍♀️ 💂🏽‍♀️ 💂🏾‍♀️ 💂🏿‍♀️ 👷‍♂️ 👷🏻‍♂️ 👷🏼‍♂️ 👷🏽‍♂️ 👷🏾‍♂️ 👷🏿‍♂️ 👷‍♀️ 👷🏻‍♀️ 👷🏼‍♀️ 👷🏽‍♀️ 👷🏾‍♀️ 👷🏿‍♀️ 👳‍♂️ 👳🏻‍♂️ 👳🏼‍♂️ 👳🏽‍♂️"},{"id":"6","event":null,"data":"निश्च\nित रूप से किसी को ब्लूबेरी जैम खरीदने में सक्षम होना चाहिए? 👳🏾‍♂️ 👳🏿‍♂️ 👳‍♀️ 👳🏻‍♀️ 👳🏼‍♀️ 👳🏽‍♀️ 👳🏾‍♀️ 👳🏿‍♀️ 🤵‍♂️ 🤵🏻‍♂️ 🤵🏼‍♂️ 🤵🏽‍♂️ 🤵🏾‍♂️ 🤵🏿‍♂️ 🤵‍♀️ 🤵🏻‍♀️ 🤵🏼‍♀️ 🤵🏽‍♀️ 🤵🏾‍♀️ 🤵🏿‍♀️ 👰‍♂️ 👰🏻‍♂️ 👰🏼‍♂️ 👰🏽‍♂️ 👰🏾‍♂️ 👰🏿‍♂️ 👰‍♀️ 👰🏻‍♀️ 👰🏼‍♀️ 👰🏽‍♀️ 👰🏾‍♀️ 👰🏿‍♀️ 👩‍🍼 👩🏻‍🍼 👩🏼‍🍼 👩🏽‍🍼 👩🏾‍🍼 👩🏿‍🍼 👨‍🍼 👨🏻‍🍼 👨🏼‍🍼 👨🏽‍🍼 👨🏾‍🍼 👨🏿‍🍼 🧑‍🍼 🧑🏻‍🍼 🧑🏼‍🍼 🧑🏽‍🍼 🧑🏾‍🍼 🧑🏿‍🍼 🧑‍🎄 🧑🏻‍🎄 🧑🏼‍🎄 🧑🏽‍🎄 🧑🏾‍🎄 🧑🏿‍🎄 🦸‍♂️ 🦸🏻‍♂️ 🦸🏼‍♂️ 🦸🏽‍♂️ 🦸🏾‍♂️ 🦸🏿‍♂️ 🦸‍♀️ 🦸🏻‍♀️ 🦸🏼‍♀️ 🦸🏽‍♀️ 🦸🏾‍♀️ 🦸🏿‍♀️ 🦹‍♂️ 🦹🏻‍♂️ 🦹🏼‍♂️ 🦹🏽‍♂️ 🦹🏾‍♂️ 🦹🏿‍♂️ 🦹‍♀️ 🦹🏻‍♀️ 🦹🏼‍♀️ 🦹🏽‍♀️ 🦹🏾‍♀️ 🦹🏿‍♀️ 🧙‍♂️ 🧙🏻‍♂️ 🧙🏼‍♂️ 🧙🏽‍♂️ 🧙🏾‍♂️ 🧙🏿‍♂️ 🧙‍♀️ 🧙🏻‍♀️ 🧙🏼‍♀️ 🧙🏽‍♀️ 🧙🏾‍♀️ 🧙🏿‍♀️ 🧚‍♂️ 🧚🏻‍♂️ 🧚🏼‍♂️ 🧚🏽‍♂️ 🧚🏾‍♂️"},{"id":"7","event":null,"data":"Bê guman pêdivî ye ku meriv karibe jama şînê peyda bike? 🧚🏿‍♂️ 🧚‍♀️ 🧚🏻‍♀️ 🧚🏼‍♀️ 🧚🏽‍♀️ 🧚🏾‍♀️ 🧚🏿‍♀️ 🧛‍♂️ 🧛🏻‍♂️ 🧛🏼‍♂️ 🧛🏽‍♂️ 🧛🏾‍♂️ 🧛🏿‍♂️ 🧛‍♀️ 🧛🏻‍♀️ 🧛🏼‍♀️ 🧛🏽‍♀️ 🧛🏾‍♀️ 🧛🏿‍♀️ 🧜‍♂️ 🧜🏻‍♂️ 🧜🏼‍♂️ 🧜🏽‍♂️ 🧜🏾‍♂️ 🧜🏿‍♂️ 🧜‍♀️ 🧜🏻‍♀️ 🧜🏼‍♀️ 🧜🏽‍♀️ 🧜🏾‍♀️ 🧜🏿‍♀️ 🧝‍♂️ 🧝🏻‍♂️ 🧝🏼‍♂️ 🧝🏽‍♂️ 🧝🏾‍♂️ 🧝🏿‍♂️ 🧝‍♀️ 🧝🏻‍♀️ 🧝🏼‍♀️ 🧝🏽‍♀️ 🧝🏾‍♀️ 🧝🏿‍♀️ 🧞‍♂️ 🧞‍♀️ 🧟‍♂️ 🧟‍♀️ 💆‍♂️ 💆🏻‍♂️ 💆🏼‍♂️ 💆🏽‍♂️ 💆🏾‍♂️ 💆🏿‍♂️ 💆‍♀️ 💆🏻‍♀️ 💆🏼‍♀️ 💆🏽‍♀️ 💆🏾‍♀️ 💆🏿‍♀️ 💇‍♂️ 💇🏻‍♂️ 💇🏼‍♂️ 💇🏽‍♂️ 💇🏾‍♂️ 💇🏿‍♂️ 💇‍♀️ 💇🏻‍♀️ 💇🏼‍♀️ 💇🏽‍♀️ 💇🏾‍♀️ 💇🏿‍♀️ 🚶‍♂️ 🚶🏻‍♂️ 🚶🏼‍♂️ 🚶🏽‍♂️ 🚶🏾‍♂️ 🚶🏿‍♂️ 🚶‍♀️ 🚶🏻‍♀️ 🚶🏼‍♀️ 🚶🏽‍♀️ 🚶🏾‍♀️ 🚶🏿‍♀️ 🧍‍♂️ 🧍🏻‍♂️ 🧍🏼‍♂️ 🧍🏽‍♂️ 🧍🏾‍♂️ 🧍🏿‍♂️ 🧍‍♀️ 🧍🏻‍♀️ 🧍🏼‍♀️ 🧍🏽‍♀️ 🧍🏾‍♀️ 🧍🏿‍♀️ 🧎‍♂️ 🧎🏻‍♂️"},{"id":"8","event":null,"data":"ប្រាក\nដណាស់ មនុស្សម្នាក់ត្រូវតែអាចទិញយៈសាពូនមី blueberry បានទេ? 🧎🏼‍♂️ 🧎🏽‍♂️ 🧎🏾‍♂️ 🧎🏿‍♂️ 🧎‍♀️ 🧎🏻‍♀️ 🧎🏼‍♀️ 🧎🏽‍♀️ 🧎🏾‍♀️ 🧎🏿‍♀️ 🧑‍🦯 🧑🏻‍🦯 🧑🏼‍🦯 🧑🏽‍🦯 🧑🏾‍🦯 🧑🏿‍🦯 👨‍🦯 👨🏻‍🦯 👨🏼‍🦯 👨🏽‍🦯 👨🏾‍🦯 👨🏿‍🦯 👩‍🦯 👩🏻‍🦯 👩🏼‍🦯 👩🏽‍🦯 👩🏾‍🦯 👩🏿‍🦯 🧑‍🦼 🧑🏻‍🦼 🧑🏼‍🦼 🧑🏽‍🦼 🧑🏾‍🦼 🧑🏿‍🦼 👨‍🦼 👨🏻‍🦼 👨🏼‍🦼 👨🏽‍🦼 👨🏾‍🦼 👨🏿‍🦼 👩‍🦼 👩🏻‍🦼 👩🏼‍🦼 👩🏽‍🦼 👩🏾‍🦼 👩🏿‍🦼 🧑‍🦽 🧑🏻‍🦽 🧑🏼‍🦽 🧑🏽‍🦽 🧑🏾‍🦽 🧑🏿‍🦽 👨‍🦽 👨🏻‍🦽 👨🏼‍🦽 👨🏽‍🦽 👨🏾‍🦽 👨🏿‍🦽 👩‍🦽 👩🏻‍🦽 👩🏼‍🦽 👩🏽‍🦽 👩🏾‍🦽 👩🏿‍🦽 🏃‍♂️ 🏃🏻‍♂️ 🏃🏼‍♂️ 🏃🏽‍♂️ 🏃🏾‍♂️ 🏃🏿‍♂️ 🏃‍♀️ 🏃🏻‍♀️ 🏃🏼‍♀️ 🏃🏽‍♀️ 🏃🏾‍♀️ 🏃🏿‍♀️ 👯‍♂️ 👯‍♀️ 🧖‍♂️ 🧖🏻‍♂️ 🧖🏼‍♂️ 🧖🏽‍♂️ 🧖🏾‍♂️ 🧖🏿‍♂️ 🧖‍♀️ 🧖🏻‍♀️ 🧖🏼‍♀️ 🧖🏽‍♀️ 🧖🏾‍♀️ 🧖🏿‍♀️ 🧗‍♂️ 🧗🏻‍♂️ 🧗🏼‍♂️ 🧗🏽‍♂️ 🧗🏾‍♂️ 🧗🏿‍♂️ 🧗‍♀️"},{"id":"9","event":null,"data":"ಖಂಡಿತವಾಗಿಯೂ ಒಬ್ಬರು ಬ್ಲೂಬೆರ್ರಿ ಜಾಮ್ ಅನ್ನು ಸಂಗ್ರಹಿಸಲು ಶಕ್ತರಾಗಿರಬೇಕು? 🧗🏻‍♀️ 🧗🏼‍♀️ 🧗🏽‍♀️ 🧗🏾‍♀️ 🧗🏿‍♀️ 🏌️‍♂️ 🏌🏻‍♂️ 🏌🏼‍♂️ 🏌🏽‍♂️ 🏌🏾‍♂️ 🏌🏿‍♂️ 🏌️‍♀️ 🏌🏻‍♀️ 🏌🏼‍♀️ 🏌🏽‍♀️ 🏌🏾‍♀️ 🏌🏿‍♀️ 🏄‍♂️ 🏄🏻‍♂️ 🏄🏼‍♂️ 🏄🏽‍♂️ 🏄🏾‍♂️ 🏄🏿‍♂️ 🏄‍♀️ 🏄🏻‍♀️ 🏄🏼‍♀️ 🏄🏽‍♀️ 🏄🏾‍♀️ 🏄🏿‍♀️ 🚣‍♂️ 🚣🏻‍♂️ 🚣🏼‍♂️ 🚣🏽‍♂️ 🚣🏾‍♂️ 🚣🏿‍♂️ 🚣‍♀️ 🚣🏻‍♀️ 🚣🏼‍♀️ 🚣🏽‍♀️ 🚣🏾‍♀️ 🚣🏿‍♀️ 🏊‍♂️ 🏊🏻‍♂️ 🏊🏼‍♂️ 🏊🏽‍♂️ 🏊🏾‍♂️ 🏊🏿‍♂️ 🏊‍♀️ 🏊🏻‍♀️ 🏊🏼‍♀️ 🏊🏽‍♀️ 🏊🏾‍♀️ 🏊🏿‍♀️ ⛹️‍♂️ ⛹🏻‍♂️ ⛹🏼‍♂️ ⛹🏽‍♂️ ⛹🏾‍♂️ ⛹🏿‍♂️ ⛹️‍♀️ ⛹🏻‍♀️ ⛹🏼‍♀️ ⛹🏽‍♀️ ⛹🏾‍♀️ ⛹🏿‍♀️ 🏋️‍♂️ 🏋🏻‍♂️ 🏋🏼‍♂️ 🏋🏽‍♂️ 🏋🏾‍♂️ 🏋🏿‍♂️ 🏋️‍♀️ 🏋🏻‍♀️ 🏋🏼‍♀️ 🏋🏽‍♀️ 🏋🏾‍♀️ 🏋🏿‍♀️ 🚴‍♂️ 🚴🏻‍♂️ 🚴🏼‍♂️ 🚴🏽‍♂️ 🚴🏾‍♂️ 🚴🏿‍♂️ 🚴‍♀️ 🚴🏻‍♀️ 🚴🏼‍♀️ 🚴🏽‍♀️ 🚴🏾‍♀️ 🚴🏿‍♀️ 🚵‍♂️ 🚵🏻‍♂️ 🚵🏼‍♂️ 🚵🏽‍♂️ 🚵🏾‍♂️ 🚵🏿‍♂️ 🚵‍♀️ 🚵🏻‍♀️"},{"id":"10","event":null,"data":"確かにブル\nーベリージャムを調達できなければなりませんか? 🚵🏼‍♀️ 🚵🏽‍♀️ 🚵🏾‍♀️ 🚵🏿‍♀️ 🤸‍♂️ 🤸🏻‍♂️ 🤸🏼‍♂️ 🤸🏽‍♂️ 🤸🏾‍♂️ 🤸🏿‍♂️ 🤸‍♀️ 🤸🏻‍♀️ 🤸🏼‍♀️ 🤸🏽‍♀️ 🤸🏾‍♀️ 🤸🏿‍♀️ 🤼‍♂️ 🤼‍♀️ 🤽‍♂️ 🤽🏻‍♂️ 🤽🏼‍♂️ 🤽🏽‍♂️ 🤽🏾‍♂️ 🤽🏿‍♂️ 🤽‍♀️ 🤽🏻‍♀️ 🤽🏼‍♀️ 🤽🏽‍♀️ 🤽🏾‍♀️ 🤽🏿‍♀️ 🤾‍♂️ 🤾🏻‍♂️ 🤾🏼‍♂️ 🤾🏽‍♂️ 🤾🏾‍♂️ 🤾🏿‍♂️ 🤾‍♀️ 🤾🏻‍♀️ 🤾🏼‍♀️ 🤾🏽‍♀️ 🤾🏾‍♀️ 🤾🏿‍♀️ 🤹‍♂️ 🤹🏻‍♂️ 🤹🏼‍♂️ 🤹🏽‍♂️ 🤹🏾‍♂️ 🤹🏿‍♂️ 🤹‍♀️ 🤹🏻‍♀️ 🤹🏼‍♀️ 🤹🏽‍♀️ 🤹🏾‍♀️ 🤹🏿‍♀️ 🧘‍♂️ 🧘🏻‍♂️ 🧘🏼‍♂️ 🧘🏽‍♂️ 🧘🏾‍♂️ 🧘🏿‍♂️ 🧘‍♀️ 🧘🏻‍♀️ 🧘🏼‍♀️ 🧘🏽‍♀️ 🧘🏾‍♀️ 🧘🏿‍♀️ 🧑‍🤝‍🧑 🧑🏻‍🤝‍🧑🏻 🧑🏻‍🤝‍🧑🏼 🧑🏻‍🤝‍🧑🏽 🧑🏻‍🤝‍🧑🏾 🧑🏻‍🤝‍🧑🏿 🧑🏼‍🤝‍🧑🏻 🧑🏼‍🤝‍🧑🏼 🧑🏼‍🤝‍🧑🏽 🧑🏼‍🤝‍🧑🏾 🧑🏼‍🤝‍🧑🏿 🧑🏽‍🤝‍🧑🏻 🧑🏽‍🤝‍🧑🏼 🧑🏽‍🤝‍🧑🏽 🧑🏽‍🤝‍🧑🏾 🧑🏽‍🤝‍🧑🏿 🧑🏾‍🤝‍🧑🏻 🧑🏾‍🤝‍🧑🏼 🧑🏾‍🤝‍🧑🏽 🧑🏾‍🤝‍🧑🏾 🧑🏾‍🤝‍🧑🏿 🧑🏿‍🤝‍🧑🏻 🧑🏿‍🤝‍🧑🏼 🧑🏿‍🤝‍🧑🏽 🧑🏿‍🤝‍🧑🏾 🧑🏿‍🤝‍🧑🏿 👩🏻‍🤝‍👩🏼 👩🏻‍🤝‍👩🏽 👩🏻‍🤝‍👩🏾 👩🏻‍🤝‍👩🏿 👩🏼‍🤝‍👩🏻"},{"id":"11","event":null,"data":"Невже треба вміти добути варення з чорниці? 👩🏼‍🤝‍👩🏽 👩🏼‍🤝‍👩🏾 👩🏼‍🤝‍👩🏿 👩🏽‍🤝‍👩🏻 👩🏽‍🤝‍👩🏼 👩🏽‍🤝‍👩🏾 👩🏽‍🤝‍👩🏿 👩🏾‍🤝‍👩🏻 👩🏾‍🤝‍👩🏼 👩🏾‍🤝‍👩🏽 👩🏾‍🤝‍👩🏿 👩🏿‍🤝‍👩🏻 👩🏿‍🤝‍👩🏼 👩🏿‍🤝‍👩🏽 👩🏿‍🤝‍👩🏾 👩🏻‍🤝‍👨🏼 👩🏻‍🤝‍👨🏽 👩🏻‍🤝‍👨🏾 👩🏻‍🤝‍👨🏿 👩🏼‍🤝‍👨🏻 👩🏼‍🤝‍👨🏽 👩🏼‍🤝‍👨🏾 👩🏼‍🤝‍👨🏿 👩🏽‍🤝‍👨🏻 👩🏽‍🤝‍👨🏼 👩🏽‍🤝‍👨🏾 👩🏽‍🤝‍👨🏿 👩🏾‍🤝‍👨🏻 👩🏾‍🤝‍👨🏼 👩🏾‍🤝‍👨🏽 👩🏾‍🤝‍👨🏿 👩🏿‍🤝‍👨🏻 👩🏿‍🤝‍👨🏼 👩🏿‍🤝‍👨🏽 👩🏿‍🤝‍👨🏾 👨🏻‍🤝‍👨🏼 👨🏻‍🤝‍👨🏽 👨🏻‍🤝‍👨🏾 👨🏻‍🤝‍👨🏿 👨🏼‍🤝‍👨🏻 👨🏼‍🤝‍👨🏽 👨🏼‍🤝‍👨🏾 👨🏼‍🤝‍👨🏿 👨🏽‍🤝‍👨🏻 👨🏽‍🤝‍👨🏼 👨🏽‍🤝‍👨🏾 👨🏽‍🤝‍👨🏿 👨🏾‍🤝‍👨🏻 👨🏾‍🤝‍👨🏼 👨🏾‍🤝‍👨🏽 👨🏾‍🤝‍👨🏿 👨🏿‍🤝‍👨🏻 👨🏿‍🤝‍👨🏼 👨🏿‍🤝‍👨🏽 👨🏿‍🤝‍👨🏾 💏 🧑🏻‍❤️‍💋‍🧑🏼 🧑🏻‍❤️‍💋‍🧑🏽 🧑🏻‍❤️‍💋‍🧑🏾 🧑🏻‍❤️‍💋‍🧑🏿 🧑🏼‍❤️‍💋‍🧑🏻 🧑🏼‍❤️‍💋‍🧑🏽 🧑🏼‍❤️‍💋‍🧑🏾 🧑🏼‍❤️‍💋‍🧑🏿 🧑🏽‍❤️‍💋‍🧑🏻 🧑🏽‍❤️‍💋‍🧑🏼 🧑🏽‍❤️‍💋‍🧑🏾 🧑🏽‍❤️‍💋‍🧑🏿 🧑🏾‍❤️‍💋‍🧑🏻 🧑🏾‍❤️‍💋‍🧑🏼 🧑🏾‍❤️‍💋‍🧑🏽 🧑🏾‍❤️‍💋‍🧑🏿 🧑🏿‍❤️‍💋‍🧑🏻 🧑🏿‍❤️‍💋‍🧑🏼 🧑🏿‍❤️‍💋‍🧑🏽 🧑🏿‍❤️‍💋‍🧑🏾 👩‍❤️‍💋‍👨 👩🏻‍❤️‍💋‍👨🏻 👩🏻‍❤️‍💋‍👨🏼 👩🏻‍❤️‍💋‍👨🏽 👩🏻‍❤️‍💋‍👨🏾 👩🏻‍❤️‍💋‍👨🏿 👩🏼‍❤️‍💋‍👨🏻 👩🏼‍❤️‍💋‍👨🏼 👩🏼‍❤️‍💋‍👨🏽 👩🏼‍❤️‍💋‍👨🏾 👩🏼‍❤️‍💋‍👨🏿 👩🏽‍❤️‍💋‍👨🏻 👩🏽‍❤️‍💋‍👨🏼 👩🏽‍❤️‍💋‍👨🏽 👩🏽‍❤️‍💋‍👨🏾 👩🏽‍❤️‍💋‍👨🏿 👩🏾‍❤️‍💋‍👨🏻 👩🏾‍❤️‍💋‍👨🏼 👩🏾‍❤️‍💋‍👨🏽 👩🏾‍❤️‍💋‍👨🏾 👩🏾‍❤️‍💋‍👨🏿"},{"id":"12","event":null,"data":"แน่นอ\nนว่าจะต้องสามารถจัดหาแยมบลูเบอร์รี่ได้? 👩🏿‍❤️‍💋‍👨🏻 👩🏿‍❤️‍💋‍👨🏼 👩🏿‍❤️‍💋‍👨🏽 👩🏿‍❤️‍💋‍👨🏾 👩🏿‍❤️‍💋‍👨🏿 👨‍❤️‍💋‍👨 👨🏻‍❤️‍💋‍👨🏻 👨🏻‍❤️‍💋‍👨🏼 👨🏻‍❤️‍💋‍👨🏽 👨🏻‍❤️‍💋‍👨🏾 👨🏻‍❤️‍💋‍👨🏿 👨🏼‍❤️‍💋‍👨🏻 👨🏼‍❤️‍💋‍👨🏼 👨🏼‍❤️‍💋‍👨🏽 👨🏼‍❤️‍💋‍👨🏾 👨🏼‍❤️‍💋‍👨🏿 👨🏽‍❤️‍💋‍👨🏻 👨🏽‍❤️‍💋‍👨🏼 👨🏽‍❤️‍💋‍👨🏽 👨🏽‍❤️‍💋‍👨🏾 👨🏽‍❤️‍💋‍👨🏿 👨🏾‍❤️‍💋‍👨🏻 👨🏾‍❤️‍💋‍👨🏼 👨🏾‍❤️‍💋‍👨🏽 👨🏾‍❤️‍💋‍👨🏾 👨🏾‍❤️‍💋‍👨🏿 👨🏿‍❤️‍💋‍👨🏻 👨🏿‍❤️‍💋‍👨🏼 👨🏿‍❤️‍💋‍👨🏽 👨🏿‍❤️‍💋‍👨🏾 👨🏿‍❤️‍💋‍👨🏿 👩‍❤️‍💋‍👩 👩🏻‍❤️‍💋‍👩🏻 👩🏻‍❤️‍💋‍👩🏼 👩🏻‍❤️‍💋‍👩🏽 👩🏻‍❤️‍💋‍👩🏾 👩🏻‍❤️‍💋‍👩🏿 👩🏼‍❤️‍💋‍👩🏻 👩🏼‍❤️‍💋‍👩🏼 👩🏼‍❤️‍💋‍👩🏽 👩🏼‍❤️‍💋‍👩🏾 👩🏼‍❤️‍💋‍👩🏿 👩🏽‍❤️‍💋‍👩🏻 👩🏽‍❤️‍💋‍👩🏼 👩🏽‍❤️‍💋‍👩🏽 👩🏽‍❤️‍💋‍👩🏾 👩🏽‍❤️‍💋‍👩🏿 👩🏾‍❤️‍💋‍👩🏻 👩🏾‍❤️‍💋‍👩🏼 👩🏾‍❤️‍💋‍👩🏽 👩🏾‍❤️‍💋‍👩🏾 👩🏾‍❤️‍💋‍👩🏿 👩🏿‍❤️‍💋‍👩🏻 👩🏿‍❤️‍💋‍👩🏼 👩🏿‍❤️‍💋‍👩🏽 👩🏿‍❤️‍💋‍👩🏾 👩🏿‍❤️‍💋‍👩🏿 💑 🧑🏻‍❤️‍🧑🏼 🧑🏻‍❤️‍🧑🏽 🧑🏻‍❤️‍🧑🏾 🧑🏻‍❤️‍🧑🏿 🧑🏼‍❤️‍🧑🏻 🧑🏼‍❤️‍🧑🏽 🧑🏼‍❤️‍🧑🏾 🧑🏼‍❤️‍🧑🏿 🧑🏽‍❤️‍🧑🏻 🧑🏽‍❤️‍🧑🏼 🧑🏽‍❤️‍🧑🏾 🧑🏽‍❤️‍🧑🏿 🧑🏾‍❤️‍🧑🏻 🧑🏾‍❤️‍🧑🏼 🧑🏾‍❤️‍🧑🏽 🧑🏾‍❤️‍🧑🏿 🧑🏿‍❤️‍🧑🏻 🧑🏿‍❤️‍🧑🏼 🧑🏿‍❤️‍🧑🏽 🧑🏿‍❤️‍🧑🏾 👩‍❤️‍👨 👩🏻‍❤️‍👨🏻 👩🏻‍❤️‍👨🏼 👩🏻‍❤️‍👨🏽 👩🏻‍❤️‍👨🏾 👩🏻‍❤️‍👨🏿 👩🏼‍❤️‍👨🏻 👩🏼‍❤️‍👨🏼 👩🏼‍❤️‍👨🏽 👩🏼‍❤️‍👨🏾 👩🏼‍❤️‍👨🏿 👩🏽‍❤️‍👨🏻 👩🏽‍❤️‍👨🏼 👩🏽‍❤️‍👨🏽 👩🏽‍❤️‍👨🏾 👩🏽‍❤️‍👨🏿 👩🏾‍❤️‍👨🏻 👩🏾‍❤️‍👨🏼 👩🏾‍❤️‍👨🏽"},{"id":"13","event":null,"data":"Сигурно некој мора да може да набави џем од боровинки? 👩🏾‍❤️‍👨🏾 👩🏾‍❤️‍👨🏿 👩🏿‍❤️‍👨🏻 👩🏿‍❤️‍👨🏼 👩🏿‍❤️‍👨🏽 👩🏿‍❤️‍👨🏾 👩🏿‍❤️‍👨🏿 👨‍❤️‍👨 👨🏻‍❤️‍👨🏻 👨🏻‍❤️‍👨🏼 👨🏻‍❤️‍👨🏽 👨🏻‍❤️‍👨🏾 👨🏻‍❤️‍👨🏿 👨🏼‍❤️‍👨🏻 👨🏼‍❤️‍👨🏼 👨🏼‍❤️‍👨🏽 👨🏼‍❤️‍👨🏾 👨🏼‍❤️‍👨🏿 👨🏽‍❤️‍👨🏻 👨🏽‍❤️‍👨🏼 👨🏽‍❤️‍👨🏽 👨🏽‍❤️‍👨🏾 👨🏽‍❤️‍👨🏿 👨🏾‍❤️‍👨🏻 👨🏾‍❤️‍👨🏼 👨🏾‍❤️‍👨🏽 👨🏾‍❤️‍👨🏾 👨🏾‍❤️‍👨🏿 👨🏿‍❤️‍👨🏻 👨🏿‍❤️‍👨🏼 👨🏿‍❤️‍👨🏽 👨🏿‍❤️‍👨🏾 👨🏿‍❤️‍👨🏿 👩‍❤️‍👩 👩🏻‍❤️‍👩🏻 👩🏻‍❤️‍👩🏼 👩🏻‍❤️‍👩🏽 👩🏻‍❤️‍👩🏾 👩🏻‍❤️‍👩🏿 👩🏼‍❤️‍👩🏻 👩🏼‍❤️‍👩🏼 👩🏼‍❤️‍👩🏽 👩🏼‍❤️‍👩🏾 👩🏼‍❤️‍👩🏿 👩🏽‍❤️‍👩🏻 👩🏽‍❤️‍👩🏼 👩🏽‍❤️‍👩🏽 👩🏽‍❤️‍👩🏾 👩🏽‍❤️‍👩🏿 👩🏾‍❤️‍👩🏻 👩🏾‍❤️‍👩🏼 👩🏾‍❤️‍👩🏽 👩🏾‍❤️‍👩🏾 👩🏾‍❤️‍👩🏿 👩🏿‍❤️‍👩🏻 👩🏿‍❤️‍👩🏼 👩🏿‍❤️‍👩🏽 👩🏿‍❤️‍👩🏾 👩🏿‍❤️‍👩🏿 👪 👨‍👩‍👦 👨‍👩‍👧 👨‍👩‍👧‍👦 👨‍👩‍👦‍👦 👨‍👩‍👧‍👧 👨‍👨‍👦 👨‍👨‍👧 👨‍👨‍👧‍👦 👨‍👨‍👦‍👦 👨‍👨‍👧‍👧 👩‍👩‍👦 👩‍👩‍👧 👩‍👩‍👧‍👦 👩‍👩‍👦‍👦 👩‍👩‍👧‍👧 👨‍👦 👨‍👦‍👦 👨‍👧 👨‍👧‍👦 👨‍👧‍👧 👩‍👦 👩‍👦‍👦 👩‍👧 👩‍👧‍👦 👩‍👧‍👧 🐕‍🦺 🐈‍⬛ 🐻‍❄️ 🏳️‍🌈 🏳️‍⚧️ 🏴‍☠️"},{"id":null,"event":"done","data":"✔"}]} \ No newline at end of file diff --git a/aimux-stream/tests/ndjson_test.rs b/aimux-stream/tests/ndjson_test.rs deleted file mode 100644 index 88e75e1a..00000000 --- a/aimux-stream/tests/ndjson_test.rs +++ /dev/null @@ -1,98 +0,0 @@ -//! NDJSON stream decoder tests — cross-chunk UTF-8 reassembly, invalid-UTF-8 -//! rejection, and bounded-buffer behavior. - -use aimux_stream::{NdjsonError, NdjsonStream}; -use bytes::Bytes; -use futures::stream::{self, StreamExt}; -use serde::Deserialize; - -#[derive(Debug, Deserialize, PartialEq)] -struct Line { - text: String, -} - -/// Feed raw byte chunks into an [`NdjsonStream`] and collect the results. -async fn collect_lines(chunks: Vec>) -> Vec> { - let items: Vec> = - chunks.into_iter().map(|c| Ok(Bytes::from(c))).collect(); - let stream = NdjsonStream::new(stream::iter(items)); - stream.collect::>().await -} - -// ── cross-chunk UTF-8 reassembly (P0-03) ───────────────────────────────── - -#[tokio::test] -async fn emoji_ndjson_line_split_at_every_byte_boundary() { - // An NDJSON line whose value contains multi-byte emoji. Splitting the full - // line at *every* byte position must still reassemble to the intact text. - // The old per-chunk `String::from_utf8_lossy` decoder would corrupt a code - // point split across two chunks into two U+FFFDs. - let payload = "{\"text\":\"😀👋\"}\n"; - let bytes = payload.as_bytes(); - for split in 1..bytes.len() { - let (a, b) = bytes.split_at(split); - let lines = collect_lines(vec![a.to_vec(), b.to_vec()]).await; - assert_eq!( - lines.len(), - 1, - "split at byte {} produced {} results", - split, - lines.len() - ); - let line = lines[0] - .as_ref() - .unwrap_or_else(|e| panic!("split at byte {split} errored: {e:?}")); - assert_eq!( - line.text, "😀👋", - "split at byte {split} corrupted the text" - ); - } -} - -#[tokio::test] -async fn invalid_utf8_line_returns_utf8_error() { - // `{"text":"` followed by 0xF0 0x9F — the first two bytes of 😀 - // (U+1F600 = F0 9F 98 80) without the trailing bytes — is an incomplete - // 4-byte sequence. Strict decoding of the complete line surfaces a `Utf8` - // error instead of silently producing replacement chars. - let chunk = vec![ - b'{', b'"', b't', b'e', b'x', b't', b'"', b':', b'"', 0xF0, 0x9F, b'\n', - ]; - let lines = collect_lines(vec![chunk]).await; - assert_eq!(lines.len(), 1, "expected exactly one result"); - match &lines[0] { - Err(NdjsonError::Utf8(_)) => {} - other => panic!("expected NdjsonError::Utf8, got {other:?}"), - } -} - -// ── bounded buffers (P1-10) ───────────────────────────────────────────── - -#[tokio::test] -async fn oversized_line_returns_line_too_large() { - // A complete line longer than the 5-byte limit is rejected. - let stream = NdjsonStream::with_max_line_size( - stream::iter(vec![Ok::<_, std::io::Error>(Bytes::from_static( - b"{\"text\":\"hi\"}\n", - ))]), - 5, - ); - let results: Vec> = stream.collect().await; - assert_eq!(results.len(), 1); - assert!(matches!(results[0], Err(NdjsonError::LineTooLarge))); -} - -#[tokio::test] -async fn buffer_growing_past_limit_without_newline_returns_line_too_large() { - // No newline ever arrives, so the buffer would grow unboundedly; the limit - // trips instead (previously this allocated forever). - let stream = NdjsonStream::with_max_line_size( - stream::iter(vec![Ok::<_, std::io::Error>(Bytes::from_static( - b"{\"text\":\"no newline here\"", - ))]), - 5, - ); - let results: Vec> = stream.collect().await; - assert_eq!(results.len(), 1); - assert!(matches!(results[0], Err(NdjsonError::LineTooLarge))); -} diff --git a/aimux-stream/tests/sse_test.rs b/aimux-stream/tests/sse_test.rs index 65e8f344..2672a962 100644 --- a/aimux-stream/tests/sse_test.rs +++ b/aimux-stream/tests/sse_test.rs @@ -1,355 +1,778 @@ -//! Independent SSE parsing tests. +//! Port of the `eventsource-parser` test suite against [`aimux_stream::SseStream`]. //! -//! The TS SDK has no standalone SSE parser tests — `parseJsonEventStream` is -//! only exercised indirectly through provider tests. These tests fill that gap, -//! covering the edge cases listed in the task against [`aimux_stream::SseStream`]. +//! `eventsource-parser` is the parser behind the AI SDK's `parseJsonEventStream`. +//! This file ports `test/parse.test.ts` and `test/stream.test.ts` from v3.1.1 +//! (MIT), driven by the same fixtures (`test/fixtures.ts`; `test/multibyte.ts` +//! is checked in as `tests/fixtures/eventsource_parser_multibyte.json`). Test +//! names follow the upstream titles. //! -//! The expected behavior mirrors `eventsource-parser`'s `EventSourceParserStream` -//! (which the TS `parseJsonEventStream` pipes through): -//! - an event is dispatched only on a terminating blank line (`\n\n`/`\r\n\r\n`); -//! - an event with no `data:` line is NOT dispatched (covers comment-only, -//! `event:`/`id:`/`retry:`-only, and blank-line keep-alives); -//! - exactly one leading U+0020 SPACE after the `:` is stripped from a field -//! value (per the SSE spec); -//! - a partial event at EOF (no terminating blank line) is dropped. +//! Adaptations forced by the Rust API (`SseStream` yields whole events, has no +//! callbacks, and works on bytes): +//! - `retry` is reported on the dispatched event instead of through `onRetry`, +//! so a `retry:` block with no `data:` is dropped, and upstream's +//! `reconnect-interval` events become `SseEvent::retry`. +//! - Upstream's `onError` for an invalid `retry` value has no counterpart: the +//! value is ignored, which is what is asserted. +//! - There is no `maxBufferSize`: like `parseJsonEventStream`, the stream is +//! unbounded, so the upstream overflow cases are not ported. +//! - A leading U+FEFF is the UTF-8 BOM at byte level. Upstream's +//! "invalid byte-order mark" case feeds a decoded U+FEFF that the JS +//! `TextDecoderStream` would already have stripped, so only the "multiple +//! places" case is ported. +//! +//! Not ported (no Rust counterpart): `onComment` call counts, `reset()` +//! (3 tests), `onError` `ParseError` payloads (3 tests), the "function passed +//! to `createParser`" guard, and the `maxBufferSize` / `onError: 'terminate'` +//! stream tests (no size limit here). +//! +//! The `aimux` module holds the few tests for behaviour upstream does not have. use aimux_stream::{SseError, SseEvent, SseStream}; use bytes::Bytes; use futures::stream::{self, StreamExt}; +use serde_json::Value; +use sha2::{Digest, Sha256}; + +// ── harness ────────────────────────────────────────────────────────────── -/// Feed `chunks` (in arrival order) into a [`SseStream`] and collect all -/// emitted events. Each chunk simulates one `Result` item from a -/// real byte stream, so splitting a single event across chunks exercises the -/// cross-chunk buffering path. -async fn collect_events(chunks: Vec<&str>) -> Vec> { - let items: Vec> = chunks +type Item = Result; +type Triple<'a> = (Option<&'a str>, Option<&'a str>, &'a str); + +async fn run_bytes(chunks: Vec>) -> Vec { + let items: Vec> = + chunks.into_iter().map(|c| Ok(Bytes::from(c))).collect(); + SseStream::new(stream::iter(items)) + .collect::>() + .await +} + +async fn run(chunks: Vec) -> Vec { + run_bytes(chunks.into_iter().map(String::into_bytes).collect()).await +} + +/// Feed `chunks` and return the events; any error item fails the test. +async fn events(chunks: Vec) -> Vec { + run(chunks) + .await .into_iter() - .map(|s| Ok(Bytes::copy_from_slice(s.as_bytes()))) - .collect(); - let stream = SseStream::new(stream::iter(items)); - stream.collect::>().await + .map(|item| item.unwrap_or_else(|e| panic!("unexpected error item: {e:?}"))) + .collect() } -fn data(event: &SseEvent) -> &str { - &event.data +fn s(chunks: &[&str]) -> Vec { + chunks.iter().map(|c| (*c).to_string()).collect() } -// ── the 12 required edge cases ──────────────────────────────────────────── +/// `(id, event, data)` of each event. +fn triples(events: &[SseEvent]) -> Vec> { + events + .iter() + .map(|e| (e.id.as_deref(), e.event.as_deref(), e.data.as_str())) + .collect() +} -#[tokio::test] -async fn single_complete_event() { - let events = collect_events(vec!["data: hello\n\n"]).await; - assert_eq!(events.len(), 1); - let event = events[0].as_ref().unwrap(); - assert_eq!(data(event), "hello"); - assert!(event.event.is_none()); - assert!(event.id.is_none()); - assert!(event.retry.is_none()); +// ── `eventsource-encoder` (used by upstream's fixtures) ────────────────── + +#[derive(Default)] +struct Msg<'a> { + event: Option<&'a str>, + retry: Option, + id: Option<&'a str>, + data: Option<&'a str>, } -#[tokio::test] -async fn multiple_consecutive_events() { - let events = collect_events(vec!["data: first\n\ndata: second\n\n"]).await; - assert_eq!(events.len(), 2); - assert_eq!(data(events[0].as_ref().unwrap()), "first"); - assert_eq!(data(events[1].as_ref().unwrap()), "second"); +fn encode_data(text: &str) -> String { + let normalized = text.replace("\r\n", "\n").replace('\r', "\n"); + let lines: Vec<&str> = normalized.split('\n').collect(); + let mut out = String::new(); + for (i, line) in lines.iter().enumerate() { + out.push_str("data: "); + out.push_str(line); + out.push_str(if i + 1 == lines.len() { "\n\n" } else { "\n" }); + } + out } -#[tokio::test] -async fn event_spanning_two_chunks_half_line_split() { - // A single event whose data line arrives in two pieces. - let events = collect_events(vec!["data: hel", "lo\n\n"]).await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), "hello"); +fn encode_comment(comment: &str) -> String { + let normalized = comment.replace("\r\n", "\n").replace('\r', "\n"); + format!(": {}\n\n", normalized.replace('\n', "\n: ")) } -#[tokio::test] -async fn event_spanning_many_tiny_chunks() { - // The same event byte-split across many tiny chunks. - let events = collect_events(vec!["da", "ta: ", "wor", "ld", "\n", "\n"]).await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), "world"); +fn encode(msg: &Msg<'_>) -> String { + let mut out = String::new(); + if let Some(event) = msg.event.filter(|e| !e.is_empty()) { + out.push_str(&format!("event: {event}\n")); + } + if let Some(retry) = msg.retry { + out.push_str(&format!("retry: {retry}\n")); + } + if let Some(id) = msg.id { + out.push_str(&format!("id: {id}\n")); + } + if let Some(data) = msg.data { + out.push_str(&encode_data(data)); + } else if !out.is_empty() { + out.push_str("\n\n"); + } + out +} + +fn done() -> String { + encode(&Msg { + event: Some("done"), + data: Some("✔"), + ..Msg::default() + }) +} + +fn data_only(data: &str) -> String { + encode(&Msg { + data: Some(data), + ..Msg::default() + }) +} + +// ── fixtures (`test/fixtures.ts`) ──────────────────────────────────────── + +struct Multibyte { + lines: Vec, + emojis: Vec, + expected: Vec<(Option, Option, String)>, +} + +fn multibyte() -> Multibyte { + let root: Value = + serde_json::from_str(include_str!("fixtures/eventsource_parser_multibyte.json")).unwrap(); + let strings = |key: &str| -> Vec { + root[key] + .as_array() + .unwrap() + .iter() + .map(|v| v.as_str().unwrap().to_string()) + .collect() + }; + let optional = |v: &Value| v.as_str().map(str::to_string); + Multibyte { + lines: strings("lines"), + emojis: strings("emojis"), + expected: root["expected"] + .as_array() + .unwrap() + .iter() + .map(|e| { + ( + optional(&e["id"]), + optional(&e["event"]), + e["data"].as_str().unwrap().to_string(), + ) + }) + .collect(), + } } -#[tokio::test] -async fn multi_line_data_field() { - let events = collect_events(vec!["data: line1\ndata: line2\n\n"]).await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), "line1\nline2"); +/// Split `text` after `units` UTF-16 code units (JS `slice(0, n)` / `slice(n)`). +fn split_utf16(text: &str, units: usize) -> (&str, &str) { + let mut seen = 0; + for (index, ch) in text.char_indices() { + if seen == units { + return text.split_at(index); + } + seen += ch.len_utf16(); + assert!(seen <= units, "UTF-16 cut inside a surrogate pair"); + } + (text, "") +} + +fn multibyte_chunks(mb: &Multibyte) -> Vec { + let per_message = mb.emojis.len().div_ceil(mb.lines.len()); + let mut chunks = Vec::new(); + for (i, line) in mb.lines.iter().enumerate() { + let start = (per_message * i).min(mb.emojis.len()); + let end = (per_message * i + per_message).min(mb.emojis.len()); + let line = format!("{line} {}", mb.emojis[start..end].join(" ")); + chunks.push(format!("id: {i}\n")); + if i % 2 == 0 { + // Even lines are split into two `data:` lines. + let (head, tail) = split_utf16(&line, 5); + chunks.push(format!("data:{head}\n")); + chunks.push(format!("data:{tail}\n\n")); + } else { + chunks.push(format!("data:{line}\n\n")); + } + } + chunks.push(done()); + chunks } -#[tokio::test] -async fn event_field() { - let events = collect_events(vec!["event: message\ndata: payload\n\n"]).await; - assert_eq!(events.len(), 1); - let event = events[0].as_ref().unwrap(); - assert_eq!(event.event.as_deref(), Some("message")); - assert_eq!(data(event), "payload"); +fn iso_timestamp(i: usize) -> String { + format!("2026-09-29T05:51:{:02}.{:03}Z", i % 60, i) } -#[tokio::test] -async fn id_field() { - let events = collect_events(vec!["id: 42\ndata: payload\n\n"]).await; - assert_eq!(events.len(), 1); - let event = events[0].as_ref().unwrap(); - assert_eq!(event.id.as_deref(), Some("42")); - assert_eq!(data(event), "payload"); +/// `^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}\.\d{3}Z$` +fn is_iso_timestamp(value: &str) -> bool { + const SHAPE: &[u8] = b"dddd-dd-ddTdd:dd:dd.dddZ"; + value.len() == SHAPE.len() + && value.bytes().zip(SHAPE).all(|(b, shape)| match shape { + b'd' => b.is_ascii_digit(), + other => b == *other, + }) } +// ── parse.test.ts ──────────────────────────────────────────────────────── + #[tokio::test] -async fn retry_field() { - let events = collect_events(vec!["retry: 5000\ndata: payload\n\n"]).await; - assert_eq!(events.len(), 1); - let event = events[0].as_ref().unwrap(); - assert_eq!(event.retry, Some(5000)); - assert_eq!(data(event), "payload"); +async fn basic_unnamed_events_stream() { + let mut chunks: Vec = (0..5).map(|i| data_only(&i.to_string())).collect(); + chunks.push(done()); + let got = events(chunks).await; + assert_eq!( + triples(&got), + vec![ + (None, None, "0"), + (None, None, "1"), + (None, None, "2"), + (None, None, "3"), + (None, None, "4"), + (None, Some("done"), "✔"), + ] + ); } #[tokio::test] -async fn comment_lines_starting_with_colon_are_ignored() { - // A comment line within an event block does not affect the event. - let events = collect_events(vec![": this is a comment\ndata: hello\n\n"]).await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), "hello"); +async fn stream_of_time_event_name() { + let mut chunks: Vec = (0..5) + .map(|i| { + encode(&Msg { + event: Some("time"), + data: Some(&iso_timestamp(i)), + ..Msg::default() + }) + }) + .collect(); + chunks.push(done()); + let got = events(chunks).await; + assert_eq!(got.len(), 6); + for event in &got[..5] { + assert_eq!(event.event.as_deref(), Some("time")); + assert!(is_iso_timestamp(&event.data), "{}", event.data); + } } #[tokio::test] -async fn comment_only_event_is_not_emitted() { - // An event consisting solely of a comment has no data line → not dispatched. - let events = collect_events(vec![": just a comment\n\n"]).await; - assert!(events.is_empty()); +async fn stream_of_time_event_names_unbalanced_chunks() { + let mut chunks = Vec::new(); + let mut ids = Vec::new(); + for i in 0..30 { + let id = (100_000_000_000_u64 + i as u64 * 7919).to_string(); + let message = encode(&Msg { + id: Some(&id), + event: Some("time"), + data: Some(&iso_timestamp(i)), + ..Msg::default() + }); + // Upstream splits at a random offset; the offset is varied + // deterministically here (the message is ASCII). + let split = (i * 11 + 3) % message.len(); + chunks.push(message[..split].to_string()); + chunks.push(message[split..].to_string()); + ids.push(id); + } + let got = events(chunks).await; + assert_eq!(got.len(), 30); + for (event, id) in got.iter().zip(&ids) { + assert_eq!(event.event.as_deref(), Some("time")); + assert_eq!(event.id.as_deref(), Some(id.as_str())); + assert!(is_iso_timestamp(&event.data), "{}", event.data); + } } #[tokio::test] -async fn done_sentinel_is_emitted_as_data() { - // The parser emits `data: [DONE]` like any other event; higher layers - // (e.g. the OpenAI provider) filter the sentinel. This mirrors the TS - // `parseJsonEventStream`, which drops `[DONE]` after parsing — the SSE - // parser itself does not special-case it. - let events = collect_events(vec!["data: [DONE]\n\n"]).await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), "[DONE]"); +async fn stream_of_identified_messages_and_retry_interval() { + let chunks = (1337..1339) + .map(|id| { + let id = id.to_string(); + encode(&Msg { + event: Some("tick"), + data: Some(&id), + id: Some(&id), + retry: Some(50), + }) + }) + .collect(); + let got = events(chunks).await; + assert_eq!( + triples(&got), + vec![ + (Some("1337"), Some("tick"), "1337"), + (Some("1338"), Some("tick"), "1338"), + ] + ); + assert!(got.iter().all(|e| e.retry == Some(50))); } #[tokio::test] -async fn blank_lines_and_heartbeat_emit_nothing() { - // Bare blank lines are keep-alives / heartbeats with no data → not emitted. - let events = collect_events(vec!["\n\n\n\n"]).await; - assert!(events.is_empty()); +async fn stream_of_heartbeat_comments_unnamed_events() { + let mut chunks = Vec::new(); + for letter in ['A', 'B', 'C', 'D', 'E'] { + chunks.push(encode_comment(" ♥")); + chunks.push(data_only(&letter.to_string())); + } + chunks.push(done()); + let got = events(chunks).await; + assert_eq!( + triples(&got), + vec![ + (None, None, "A"), + (None, None, "B"), + (None, None, "C"), + (None, None, "D"), + (None, None, "E"), + (None, Some("done"), "✔"), + ] + ); } #[tokio::test] -async fn crlf_vs_lf_line_endings() { - // CRLF-terminated single event. - let events = collect_events(vec!["data: hello\r\n\r\n"]).await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), "hello"); +async fn stream_of_multi_line_data_events() { + let mut chunks = s(&[ + "event: stock\n", + "data: YHOO\n", + "data: +2\n", + "data: 10\n\n", + "event: stock\n", + "data: GOOG\n", + "data: -8\n", + "data: 1881\n\n", + ]); + chunks.push(done()); + let got = events(chunks).await; + assert_eq!( + triples(&got)[..2], + [ + (None, Some("stock"), "YHOO\n+2\n10"), + (None, Some("stock"), "GOOG\n-8\n1881"), + ] + ); } #[tokio::test] -async fn crlf_multi_line_data() { - // CRLF-terminated multi-line data joins with a single `\n`. - let events = collect_events(vec!["data: line1\r\ndata: line2\r\n\r\n"]).await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), "line1\nline2"); +async fn stream_of_multi_byte_events() { + let mb = multibyte(); + let got = events(multibyte_chunks(&mb)).await; + let got: Vec<_> = got.into_iter().map(|e| (e.id, e.event, e.data)).collect(); + assert_eq!(got, mb.expected); } #[tokio::test] -async fn incomplete_event_at_eof_without_blank_line_is_dropped() { - // No terminating blank line → the buffered partial is not a complete event - // and is dropped (matches eventsource-parser's EventSourceParserStream, - // which has no flush handler). - let events = collect_events(vec!["data: hello"]).await; - assert!(events.is_empty()); +async fn stream_of_multi_byte_events_with_some_empty_lines_thrown_in() { + let got = events(vec![ + "\n\n\n\nid: 1\ndata: 我現在都看實況不玩遊戲\n\n".to_string(), + done(), + ]) + .await; + assert_eq!( + triples(&got), + vec![ + (Some("1"), None, "我現在都看實況不玩遊戲"), + (None, Some("done"), "✔"), + ] + ); } #[tokio::test] -async fn incomplete_multi_line_event_at_eof_is_dropped() { - // A buffered data line that never receives its terminating blank line is - // dropped, even if the data line itself looks complete. - let events = collect_events(vec!["data: hello\n"]).await; - assert!(events.is_empty()); +async fn stream_of_leading_bom() { + let got = events(vec![ + "\u{FEFF}data: bomful 1\n\n".to_string(), + "data: bomless 2\n\n".to_string(), + done(), + ]) + .await; + assert_eq!( + triples(&got), + vec![ + (None, None, "bomful 1"), + (None, None, "bomless 2"), + (None, Some("done"), "✔"), + ] + ); } -// ── additional spec-faithfulness coverage ──────────────────────────────── - #[tokio::test] -async fn data_without_space_after_colon() { - // `data:hello` (no space) → value is `hello`. - let events = collect_events(vec!["data:hello\n\n"]).await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), "hello"); +async fn stream_containing_byte_order_mark_multiple_places() { + let got = events(vec![ + "\u{FEFF}data: bomful 1\n\n".to_string(), + "\u{FEFF}data: bomful 2\n\n".to_string(), + "data: bomless 3\n\n".to_string(), + done(), + ]) + .await; + // Only the BOM at the start of the stream is stripped; the second one + // turns the field name into an unknown field, so that event has no data. + assert_eq!( + triples(&got), + vec![ + (None, None, "bomful 1"), + (None, None, "bomless 3"), + (None, Some("done"), "✔"), + ] + ); } #[tokio::test] -async fn only_one_leading_space_is_removed() { - // Per the SSE spec exactly ONE leading U+0020 SPACE is removed; the rest - // of the value is preserved verbatim. - let events = collect_events(vec!["data: two leading spaces\n\n"]).await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), " two leading spaces"); +async fn stream_using_carriage_returns() { + let mut chunks = s(&[ + "data: dog\r", + "data: bark\r\r", + "data: cat\r", + "data: meow\r\r", + ]); + chunks.push(done()); + let got = events(chunks).await; + assert_eq!( + triples(&got), + vec![ + (None, None, "dog\nbark"), + (None, None, "cat\nmeow"), + (None, Some("done"), "✔"), + ] + ); } #[tokio::test] -async fn empty_data_value() { - // `data:` with nothing after it contributes an empty data line. - let events = collect_events(vec!["data:\n\n"]).await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), ""); +async fn stream_using_line_feeds() { + let mut chunks = s(&[ + "data: cow\n", + "data: moo\n\n", + "data: horse\n", + "data: neigh\n\n", + ]); + chunks.push(done()); + let got = events(chunks).await; + assert_eq!( + triples(&got), + vec![ + (None, None, "cow\nmoo"), + (None, None, "horse\nneigh"), + (None, Some("done"), "✔"), + ] + ); } #[tokio::test] -async fn event_field_without_data_is_not_emitted() { - // `event:` without a `data:` line → not dispatched. - let events = collect_events(vec!["event: ping\n\n"]).await; - assert!(events.is_empty()); +async fn stream_using_carriage_returns_and_line_feeds() { + let mut chunks = s(&[ + "data: sheep\r\n", + "data: bleat\r\n\r\n", + "data: pig\r\n", + "data: oink\r\n\r\n", + ]); + chunks.push(done()); + let got = events(chunks).await; + assert_eq!( + triples(&got), + vec![ + (None, None, "sheep\nbleat"), + (None, None, "pig\noink"), + (None, Some("done"), "✔"), + ] + ); } #[tokio::test] -async fn id_field_without_data_is_not_emitted() { - let events = collect_events(vec!["id: 99\n\n"]).await; - assert!(events.is_empty()); +async fn stream_with_varying_odd_uses_of_comments() { + let mb = multibyte(); + let mut chunks = s(&[": Hello\n\n"]); + chunks.push(":".repeat(300)); + chunks.extend(s(&[ + "\n", + "data: First\n\n", + ": Первый", + ": 第二", + "\n", + "data: Second\n\n", + ])); + chunks.extend(std::iter::repeat_n(": Moop \n".to_string(), 10)); + chunks.extend(s(&[ + ": ثالث", + "\n", + "data: Third\n\n", + ":നാലാമത്തെ", + "\n", + "data: Fourth\n\n", + ])); + chunks.push(format!(": {} :", mb.emojis[..100].join(" "))); + chunks.extend(s(&["\n", "data: Fifth\n\n"])); + chunks.push(done()); + let got = events(chunks).await; + assert_eq!( + triples(&got), + vec![ + (None, None, "First"), + (None, None, "Second"), + (None, None, "Third"), + (None, None, "Fourth"), + (None, None, "Fifth"), + (None, Some("done"), "✔"), + ] + ); } #[tokio::test] -async fn retry_field_without_data_is_not_emitted() { - let events = collect_events(vec!["retry: 1000\n\n"]).await; - assert!(events.is_empty()); +async fn stream_with_even_more_odd_uses_of_comments() { + let long = "x".repeat(2 * 1024 + 1); + let chunks = vec![ + "data:1\r\r:\0\n:\r\ndata:2\n\n:".to_string(), + long.clone(), + "\rdata:3\n\n:data:fail\r:".to_string(), + long, + "\ndata:4\n\n".to_string(), + "data:5".to_string(), + ]; + let got = events(chunks).await; + // No newline after the last message, thus not emitted. + assert_eq!( + triples(&got), + vec![ + (None, None, "1"), + (None, None, "2"), + (None, None, "3"), + (None, None, "4"), + ] + ); } #[tokio::test] -async fn event_with_all_fields_together() { - let events = collect_events(vec!["event: update\nid: 7\nretry: 3000\ndata: payload\n\n"]).await; - assert_eq!(events.len(), 1); - let e = events[0].as_ref().unwrap(); - assert_eq!(e.event.as_deref(), Some("update")); - assert_eq!(e.id.as_deref(), Some("7")); - assert_eq!(e.retry, Some(3000)); - assert_eq!(data(e), "payload"); +async fn stream_with_empty_event_field() { + let got = events(vec![ + "event:\ndata: Hello 1\n\n".to_string(), + "event:\n\n".to_string(), + done(), + ]) + .await; + assert_eq!( + triples(&got), + vec![(None, None, "Hello 1"), (None, Some("done"), "✔")] + ); } #[tokio::test] -async fn intermixed_comments_between_events() { - let events = collect_events(vec![ - ": heartbeat\ndata: first\n\n", - ": another comment\ndata: second\n\n", +async fn stream_with_empty_retry_field() { + let got = events(vec![ + encode(&Msg { + id: Some("1"), + retry: Some(500), + data: Some("🥌"), + ..Msg::default() + }), + "id:2\nretry:\ndata:🧹\n\n".to_string(), + encode(&Msg { + id: Some("3"), + data: Some("✅"), + ..Msg::default() + }), ]) .await; - assert_eq!(events.len(), 2); - assert_eq!(data(events[0].as_ref().unwrap()), "first"); - assert_eq!(data(events[1].as_ref().unwrap()), "second"); + assert_eq!( + triples(&got), + vec![ + (Some("1"), None, "🥌"), + (Some("2"), None, "🧹"), + (Some("3"), None, "✅"), + ] + ); + // The empty `retry` is ignored; `retry: 500` belongs to the first event. + assert_eq!( + got.iter().map(|e| e.retry).collect::>(), + vec![Some(500), None, None] + ); } #[tokio::test] -async fn invalid_retry_value_is_ignored() { - // A non-numeric `retry:` value is ignored; the event is still dispatched - // because it has a data line. - let events = collect_events(vec!["retry: not-a-number\ndata: payload\n\n"]).await; - assert_eq!(events.len(), 1); - let e = events[0].as_ref().unwrap(); - assert_eq!(e.retry, None); - assert_eq!(data(e), "payload"); +async fn stream_with_oddly_shaped_data_field() { + let got = events(vec![ + "data:\n\ndata\ndata\n\ndata:test\n\n".to_string(), + done(), + ]) + .await; + // `data:\n\n` dispatches an event with empty data; `data\ndata\n\n` is two + // empty data lines, joined by a newline. + assert_eq!( + triples(&got), + vec![ + (None, None, ""), + (None, None, "\n"), + (None, None, "test"), + (None, Some("done"), "✔"), + ] + ); } #[tokio::test] -async fn unknown_field_is_ignored() { - let events = collect_events(vec!["foo: bar\ndata: payload\n\n"]).await; - assert_eq!(events.len(), 1); - assert_eq!(data(events[0].as_ref().unwrap()), "payload"); +async fn stream_with_cr_separating_chunks_of_same_event() { + // A CR at the end of a chunk may be half of a CRLF, so it must not end + // the line yet: otherwise `A\nB` and `C` would be two events. + // https://github.com/rexxars/eventsource-parser/issues/17 + let got = events(s(&["data: A\r\n", "data: B\r", "\n", "data: C\r\n", "\n"])).await; + assert_eq!(triples(&got), vec![(None, None, "A\nB\nC")]); } #[tokio::test] -async fn mixed_lf_and_crlf_events() { - // An LF event followed by a CRLF event in the same stream. - let events = collect_events(vec!["data: lf\n\ndata: crlf\r\n\r\n"]).await; - assert_eq!(events.len(), 2); - assert_eq!(data(events[0].as_ref().unwrap()), "lf"); - assert_eq!(data(events[1].as_ref().unwrap()), "crlf"); +async fn stream_with_partially_incorrect_retry_fields() { + let got = events(s(&["retry:1000\nretry:2000x\ndata:x\n\n"])).await; + // `2000x` is not all ASCII digits and is ignored; `1000` is kept. + assert_eq!(triples(&got), vec![(None, None, "x")]); + assert_eq!(got[0].retry, Some(1000)); } -// ── cross-chunk UTF-8 reassembly (P0-03) ───────────────────────────────── +#[tokio::test] +async fn stream_with_id_field_containing_a_null_character() { + // An `id` containing U+0000 is ignored, so the earlier `123` survives. + let got = events(s(&["id: 123\nid: bad\0id\ndata: hello\n\n"])).await; + assert_eq!(triples(&got), vec![(Some("123"), None, "hello")]); +} -/// Like [`collect_events`] but takes raw byte chunks, so a chunk that ends in -/// the middle of a multi-byte UTF-8 sequence (not representable as a `&str`) -/// can be fed in. -async fn collect_events_bytes(chunks: Vec>) -> Vec> { - let items: Vec> = - chunks.into_iter().map(|c| Ok(Bytes::from(c))).collect(); - let stream = SseStream::new(stream::iter(items)); - stream.collect::>().await +#[tokio::test] +async fn stream_with_incorrect_retry_fields() { + let got = events(s(&[ + "\nretry: 500\n\ndata: first\n\nretry: 50x\n\ndata: second\n\n", + ])) + .await; + assert_eq!( + triples(&got), + vec![(None, None, "first"), (None, None, "second")] + ); } #[tokio::test] -async fn chinese_sse_event_split_at_every_byte_boundary() { - // A Chinese SSE event whose data contains multi-byte UTF-8. Splitting the - // full event at *every* byte position must still reassemble to the intact - // text. The old per-chunk `String::from_utf8_lossy` decoder would corrupt - // a code point split across two chunks into two U+FFFDs. - let payload = "data: 你好世界\n\n"; - let bytes = payload.as_bytes(); - for split in 1..bytes.len() { - let (a, b) = bytes.split_at(split); - let events = collect_events_bytes(vec![a.to_vec(), b.to_vec()]).await; - assert_eq!( - events.len(), - 1, - "split at byte {} produced {} results", - split, - events.len() - ); - let event = events[0] - .as_ref() - .unwrap_or_else(|e| panic!("split at byte {split} errored: {e:?}")); - assert_eq!( - event.data, "你好世界", - "split at byte {split} corrupted the data" - ); - } +async fn stream_with_unknown_fields_in_the_stream() { + let got = events(vec![ + "data:abc\n data\ndata\nfoobar:xxx\njustsometext\n:thisisacommentyay\ndata:123\n\n" + .to_string(), + done(), + ]) + .await; + assert_eq!( + triples(&got), + vec![(None, None, "abc\n\n123"), (None, Some("done"), "✔")] + ); } #[tokio::test] -async fn invalid_utf8_frame_returns_utf8_error() { - // `data: ` followed by 0xE4 0xBD — the first two bytes of `你` (U+4F60 = - // E4 BD A0) without the trailing byte — is an incomplete UTF-8 sequence. - // Strict decoding of the complete frame surfaces a `Utf8` error instead - // of silently producing a replacement char. - let chunk = vec![b'd', b'a', b't', b'a', b':', b' ', 0xE4, 0xBD, b'\n', b'\n']; - let events = collect_events_bytes(vec![chunk]).await; - assert_eq!(events.len(), 1, "expected exactly one result"); - match &events[0] { - Err(SseError::Utf8(_)) => {} - other => panic!("expected SseError::Utf8, got {other:?}"), +async fn stream_with_huge_data_chunks() { + const TEN_MEGABYTES: usize = 1024 * 1024 * 10; + const EXPECTED_SHA256: &str = + "e094a44a2436226ea9feb04e413a28de012b406012ec0eb6b37ad0a19d403660"; + + let mb = multibyte(); + let data_chunk = encode_data(&format!( + "{}\n{}", + mb.lines.join("\n\n"), + mb.emojis.join(" ") + )) + .trim() + .to_string(); + let mut chunks = s(&[": hello\n\n"]); + let mut written = 0; + while written < TEN_MEGABYTES { + chunks.push(data_chunk.clone()); + written += data_chunk.len(); } + chunks.extend(s(&["\n\n", ": END-OF-STREAM\n\n"])); + chunks.push(encode(&Msg { + event: Some("done"), + data: Some(EXPECTED_SHA256), + ..Msg::default() + })); + + let got: Vec = run(chunks).await.into_iter().map(Result::unwrap).collect(); + assert_eq!(got.len(), 2); + // JS `String.length` counts UTF-16 code units. + assert_eq!(got[0].data.encode_utf16().count(), 4_808_512); + assert_eq!( + format!("{:x}", Sha256::digest(got[0].data.as_bytes())), + got[1].data + ); } -// ── bounded buffers (P1-10) ───────────────────────────────────────────── - #[tokio::test] -async fn oversized_frame_returns_frame_too_large() { - // A complete event whose frame (`data: hello world`) exceeds the 10-byte - // limit is rejected; the frame+terminator is dropped so the stream ends. - let stream = SseStream::with_max_event_size( - stream::iter(vec![Ok::<_, std::io::Error>(Bytes::from_static( - b"data: hello world\n\n", - ))]), - 10, - ); - let results: Vec<_> = stream.collect().await; - assert_eq!(results.len(), 1); - assert!(matches!(results[0], Err(SseError::FrameTooLarge))); +async fn large_input_without_a_terminator_is_not_an_error() { + // Unbounded, as upstream: a partial block is simply never dispatched. + assert!(run(vec!["x".repeat(4 * 1024 * 1024)]).await.is_empty()); } +// ── stream.test.ts ─────────────────────────────────────────────────────── + #[tokio::test] -async fn buffer_growing_past_limit_without_terminator_returns_frame_too_large() { - // No terminator ever arrives, so the buffer would grow unboundedly; the - // limit trips instead (previously this allocated forever). - let stream = SseStream::with_max_event_size( - stream::iter(vec![Ok::<_, std::io::Error>(Bytes::from_static( - b"data: no terminator here", - ))]), - 10, - ); - let results: Vec<_> = stream.collect().await; - assert_eq!(results.len(), 1); - assert!(matches!(results[0], Err(SseError::FrameTooLarge))); +async fn can_use_event_source_parser_stream() { + let chunks = (0..10) + .map(|i| { + encode(&Msg { + event: Some("foo"), + id: Some(&format!("evt-{i}")), + data: Some(&format!("Hello {i}")), + ..Msg::default() + }) + }) + .collect::>() + .concat(); + let got = events(vec![chunks]).await; + assert_eq!(got.len(), 10); + assert_eq!(triples(&got)[0], (Some("evt-0"), Some("foo"), "Hello 0")); + assert_eq!(triples(&got)[9], (Some("evt-9"), Some("foo"), "Hello 9")); +} + +// ── aimux-specific behaviour (not covered upstream) ────────────────────── + +mod aimux { + use super::*; + + #[tokio::test] + async fn invalid_utf8_is_a_terminal_error() { + // Upstream decodes lossily (`TextDecoder`); SseStream decodes strictly + // and, like every decoder error, ends the stream after reporting it. + // `E4 BD` is the start of `你` without its last byte. + let mut bytes = b"data: ".to_vec(); + bytes.extend_from_slice(&[0xE4, 0xBD]); + bytes.extend_from_slice(b"\ndata: tail\n\ndata: next\n\n"); + let results = run_bytes(vec![bytes]).await; + assert_eq!(results.len(), 1); + assert!(matches!(results[0], Err(SseError::Utf8(_)))); + } + + #[tokio::test] + async fn multibyte_fixture_split_at_arbitrary_byte_sizes() { + // Chunks can end inside a code point or inside the BOM; the parser + // works on bytes, so the result must not depend on where they end. + let mb = multibyte(); + let mut stream_bytes = "\u{FEFF}".as_bytes().to_vec(); + stream_bytes.extend(multibyte_chunks(&mb).concat().into_bytes()); + for size in [1, 2, 3, 5, 8, 13] { + let chunks = stream_bytes.chunks(size).map(<[u8]>::to_vec).collect(); + let got: Vec<_> = run_bytes(chunks) + .await + .into_iter() + .map(|item| { + let e = item.unwrap_or_else(|err| panic!("chunk size {size}: {err:?}")); + (e.id, e.event, e.data) + }) + .collect(); + assert_eq!(got, mb.expected, "chunk size {size}"); + } + } + + #[tokio::test] + async fn transport_error_is_reported_as_a_stream_error() { + let items: Vec> = vec![ + Ok(Bytes::from_static(b"data: a\n\n")), + Err(std::io::Error::other("boom")), + Ok(Bytes::from_static(b"data: b\n\n")), + ]; + let results = SseStream::new(stream::iter(items)) + .collect::>() + .await; + // The transport error is terminal: nothing after it is read. + assert_eq!(results.len(), 2); + assert_eq!(results[0].as_ref().unwrap().data, "a"); + assert!(matches!(&results[1], Err(SseError::Stream(e)) if e.to_string() == "boom")); + } } diff --git a/aimux-stream/tests/streaming_tool_call_tracker_test.rs b/aimux-stream/tests/streaming_tool_call_tracker_test.rs deleted file mode 100644 index 25a9b765..00000000 --- a/aimux-stream/tests/streaming_tool_call_tracker_test.rs +++ /dev/null @@ -1,787 +0,0 @@ -//! Rust translation of -//! `packages/provider-utils/src/streaming-tool-call-tracker.test.ts` (18 cases). -//! -//! The TS test collects emitted parts on a shared `controller.enqueue` sink and -//! resets it with `parts.length = 0` between assertions. The Rust equivalent -//! accumulates parts inside the tracker: inspect with `parts()` and reset with -//! `clear_parts()`. - -use aimux_stream::{ - StreamingToolCallDelta, StreamingToolCallTracker, ToolCallStreamPart, TrackerError, - TypeValidation, -}; -use serde_json::{Value, json}; - -/// Convenience builder alias. -fn d() -> StreamingToolCallDelta { - StreamingToolCallDelta::default() -} - -/// A tracker with no metadata handling (`M = ()`), matching the TS tests that -/// don't exercise the metadata hooks. Pinning `M = ()` here lets type -/// inference resolve `ToolCallStreamPart::ToolCall { provider_metadata: None }` -/// in the assertions. -fn new_tracker() -> StreamingToolCallTracker<()> { - StreamingToolCallTracker::<()>::new() -} - -fn tool_call_id(part: &ToolCallStreamPart) -> &str { - match part { - ToolCallStreamPart::ToolCall { tool_call_id, .. } => tool_call_id, - _ => panic!("expected tool-call part"), - } -} - -// == processDelta ========================================================== - -#[test] -fn single_tool_call_accumulated_across_multiple_deltas() { - // TS: "should handle a single tool call accumulated across multiple deltas" - let mut tracker = new_tracker(); - - // First delta: new tool call with id and name. - tracker - .process_delta( - &d().index(0) - .id("call_1") - .tool_type("function") - .function_name("get_weather") - .arguments("{\"ci"), - ) - .unwrap(); - - assert_eq!( - tracker.parts(), - &[ - ToolCallStreamPart::ToolInputStart { - id: "call_1".into(), - tool_name: "get_weather".into(), - }, - ToolCallStreamPart::ToolInputDelta { - id: "call_1".into(), - delta: "{\"ci".into(), - }, - ] - ); - - tracker.clear_parts(); - - // Second delta: more arguments. - tracker - .process_delta(&d().index(0).arguments("ty\": \"San")) - .unwrap(); - - assert_eq!( - tracker.parts(), - &[ToolCallStreamPart::ToolInputDelta { - id: "call_1".into(), - delta: "ty\": \"San".into(), - }] - ); - - tracker.clear_parts(); - - // Third delta: completes the JSON -- must not finalize before flush, since a - // parsable buffer can still be the prefix of longer arguments. - tracker - .process_delta(&d().index(0).arguments(" Francisco\"}")) - .unwrap(); - - assert_eq!( - tracker.parts(), - &[ToolCallStreamPart::ToolInputDelta { - id: "call_1".into(), - delta: " Francisco\"}".into(), - }] - ); - - tracker.clear_parts(); - - tracker.flush(); - - assert_eq!( - tracker.parts(), - &[ - ToolCallStreamPart::ToolInputEnd { - id: "call_1".into(), - }, - ToolCallStreamPart::ToolCall { - tool_call_id: "call_1".into(), - tool_name: "get_weather".into(), - input: "{\"city\": \"San Francisco\"}".into(), - provider_metadata: None, - }, - ] - ); -} - -#[test] -fn full_tool_call_in_single_chunk() { - // TS: "should handle a full tool call in a single chunk" - let mut tracker = new_tracker(); - - tracker - .process_delta( - &d().index(0) - .id("call_1") - .tool_type("function") - .function_name("get_weather") - .arguments("{\"city\": \"London\"}"), - ) - .unwrap(); - - assert_eq!( - tracker.parts(), - &[ - ToolCallStreamPart::ToolInputStart { - id: "call_1".into(), - tool_name: "get_weather".into(), - }, - ToolCallStreamPart::ToolInputDelta { - id: "call_1".into(), - delta: "{\"city\": \"London\"}".into(), - }, - ] - ); - - tracker.clear_parts(); - tracker.flush(); - - assert_eq!( - tracker.parts(), - &[ - ToolCallStreamPart::ToolInputEnd { - id: "call_1".into(), - }, - ToolCallStreamPart::ToolCall { - tool_call_id: "call_1".into(), - tool_name: "get_weather".into(), - input: "{\"city\": \"London\"}".into(), - provider_metadata: None, - }, - ] - ); -} - -#[test] -fn not_finalize_when_argument_prefix_is_parsable_json() { - // TS: "should not finalize a tool call when its argument prefix is parsable JSON" - let mut tracker = new_tracker(); - - tracker - .process_delta( - &d().index(0) - .id("call_1") - .tool_type("function") - .function_name("search") - .arguments("{\"query\": \"test\"}"), - ) - .unwrap(); - - // The parsable prefix must not emit tool-input-end / tool-call. - let types: Vec<&str> = tracker - .parts() - .iter() - .map(|p| match p { - ToolCallStreamPart::ToolInputStart { .. } => "tool-input-start", - ToolCallStreamPart::ToolInputDelta { .. } => "tool-input-delta", - ToolCallStreamPart::ToolInputEnd { .. } => "tool-input-end", - ToolCallStreamPart::ToolCall { .. } => "tool-call", - }) - .collect(); - assert_eq!(types, vec!["tool-input-start", "tool-input-delta"]); - - tracker - .process_delta(&d().index(0).arguments(", \"limit\": 10}")) - .unwrap(); - - tracker.flush(); - - assert_eq!( - tracker.parts().last().unwrap(), - &ToolCallStreamPart::ToolCall { - tool_call_id: "call_1".into(), - tool_name: "search".into(), - input: "{\"query\": \"test\"}, \"limit\": 10}".into(), - provider_metadata: None, - } - ); - - let tool_call_count = tracker - .parts() - .iter() - .filter(|p| matches!(p, ToolCallStreamPart::ToolCall { .. })) - .count(); - assert_eq!(tool_call_count, 1); -} - -#[test] -fn multiple_concurrent_tool_calls() { - // TS: "should handle multiple concurrent tool calls" - let mut tracker = new_tracker(); - - tracker - .process_delta( - &d().index(0) - .id("call_1") - .tool_type("function") - .function_name("get_weather") - .arguments(""), - ) - .unwrap(); - tracker - .process_delta( - &d().index(1) - .id("call_2") - .tool_type("function") - .function_name("get_time") - .arguments(""), - ) - .unwrap(); - - assert_eq!( - tracker.parts(), - &[ - ToolCallStreamPart::ToolInputStart { - id: "call_1".into(), - tool_name: "get_weather".into(), - }, - ToolCallStreamPart::ToolInputStart { - id: "call_2".into(), - tool_name: "get_time".into(), - }, - ] - ); -} - -#[test] -fn skip_deltas_for_already_finished_tool_calls() { - // TS: "should skip deltas for already-finished tool calls" - let mut tracker = new_tracker(); - - tracker - .process_delta( - &d().index(0) - .id("call_1") - .tool_type("function") - .function_name("fn") - .arguments("{}"), - ) - .unwrap(); - - // Finalize via flush. - tracker.flush(); - tracker.clear_parts(); - - // Late delta for the same tool call. - tracker - .process_delta(&d().index(0).arguments("extra")) - .unwrap(); - - assert!(tracker.parts().is_empty()); -} - -#[test] -fn skip_delta_emission_when_arguments_null() { - // TS: "should skip delta emission when arguments are null". - // `arguments: null` maps to `None` (omitted on the builder); behaviorally - // identical to a `function` with no arguments field. - let mut tracker = new_tracker(); - - tracker - .process_delta( - &d().index(0) - .id("call_1") - .tool_type("function") - .function_name("fn") - .arguments(""), - ) - .unwrap(); - - tracker.clear_parts(); - - // Delta with null arguments (no `.arguments(...)` call -> None). - tracker.process_delta(&d().index(0)).unwrap(); - - assert!(tracker.parts().is_empty()); -} - -#[test] -fn use_index_fallback_when_index_not_provided() { - // TS: "should use index fallback when index is not provided" - let mut tracker = new_tracker(); - - tracker - .process_delta( - &d().id("call_1") - .tool_type("function") - .function_name("fn1") - .arguments("{}"), - ) - .unwrap(); - tracker - .process_delta( - &d().id("call_2") - .tool_type("function") - .function_name("fn2") - .arguments("{}"), - ) - .unwrap(); - - let starts: Vec<_> = tracker - .parts() - .iter() - .filter(|p| matches!(p, ToolCallStreamPart::ToolInputStart { .. })) - .cloned() - .collect(); - assert_eq!( - starts, - vec![ - ToolCallStreamPart::ToolInputStart { - id: "call_1".into(), - tool_name: "fn1".into(), - }, - ToolCallStreamPart::ToolInputStart { - id: "call_2".into(), - tool_name: "fn2".into(), - }, - ] - ); -} - -#[test] -fn generate_id_when_id_missing() { - // TS: id is missing → `toolCall.id ?? generateId()` generates a fallback id - // instead of throwing. The default generator returns "generated". - let mut tracker = new_tracker(); - tracker - .process_delta(&d().index(0).tool_type("function").function_name("fn")) - .unwrap(); - - assert_eq!( - tracker.parts(), - &[ToolCallStreamPart::ToolInputStart { - id: "generated".into(), - tool_name: "fn".into(), - }] - ); -} - -#[test] -fn throw_when_function_name_missing() { - // TS: "should throw when function.name is missing". - // `function: {}` (TS) maps to `function: None` here; both yield a missing - // name and the same error. - let mut tracker = new_tracker(); - let err = tracker - .process_delta(&d().index(0).id("call_1").tool_type("function")) - .unwrap_err(); - assert_eq!(err, TrackerError::MissingFunctionName); - assert_eq!(err.to_string(), "Expected 'function.name' to be a string."); -} - -// == typeValidation ======================================================== - -#[test] -fn no_validate_type_with_type_validation_none() { - // TS: "should not validate type with typeValidation: none" - let mut tracker = new_tracker().with_type_validation(TypeValidation::None); - // Should not throw even with a non-function type. - let result = tracker.process_delta( - &d().index(0) - .id("call_1") - .tool_type("custom") - .function_name("fn") - .arguments(""), - ); - assert!(result.is_ok()); -} - -#[test] -fn validate_type_when_present_with_type_validation_if_present() { - // TS: "should validate type when present with typeValidation: if-present" - let mut tracker = new_tracker().with_type_validation(TypeValidation::IfPresent); - - // Should throw for a non-function type. - let err = tracker - .process_delta( - &d().index(0) - .id("call_1") - .tool_type("custom") - .function_name("fn") - .arguments(""), - ) - .unwrap_err(); - assert_eq!(err, TrackerError::InvalidType); - assert_eq!(err.to_string(), "Expected 'function' type."); - - // Should not throw when type is null (absent). The first delta errored - // before creating any tool call, so index 0 is still new. - let result = - tracker.process_delta(&d().index(0).id("call_1").function_name("fn").arguments("")); - assert!(result.is_ok()); -} - -#[test] -fn require_function_type_with_type_validation_required() { - // TS: "should require function type with typeValidation: required" - let mut tracker = new_tracker().with_type_validation(TypeValidation::Required); - - // Should throw when type is null/undefined. - let err = tracker - .process_delta(&d().index(0).id("call_1").function_name("fn").arguments("")) - .unwrap_err(); - assert_eq!(err, TrackerError::InvalidType); - - // Should not throw for 'function' type. - let result = tracker.process_delta( - &d().index(0) - .id("call_1") - .tool_type("function") - .function_name("fn") - .arguments(""), - ); - assert!(result.is_ok()); -} - -// == flush ================================================================= - -#[test] -fn finalize_unfinished_tool_calls_on_flush() { - // TS: "should finalize unfinished tool calls on flush" - let mut tracker = new_tracker(); - - // Start a tool call but don't complete it. - tracker - .process_delta( - &d().index(0) - .id("call_1") - .tool_type("function") - .function_name("fn") - .arguments("{\"key\": \"val"), - ) - .unwrap(); - - tracker.clear_parts(); - - // Flush should finalize. - tracker.flush(); - - assert_eq!( - tracker.parts(), - &[ - ToolCallStreamPart::ToolInputEnd { - id: "call_1".into(), - }, - ToolCallStreamPart::ToolCall { - tool_call_id: "call_1".into(), - tool_name: "fn".into(), - input: "{\"key\": \"val".into(), - provider_metadata: None, - }, - ] - ); -} - -#[test] -fn not_refinalize_already_finished_tool_calls() { - // TS: "should not re-finalize already finished tool calls" - let mut tracker = new_tracker(); - - tracker - .process_delta( - &d().index(0) - .id("call_1") - .tool_type("function") - .function_name("fn") - .arguments("{}"), - ) - .unwrap(); - - // First flush finalizes the tool call. - tracker.flush(); - tracker.clear_parts(); - - tracker.flush(); - - // No events should be emitted since the tool call was already finished. - assert!(tracker.parts().is_empty()); -} - -// == metadata ============================================================= - -#[test] -fn extract_and_include_provider_metadata_in_tool_call_events() { - // TS: "should extract and include provider metadata in tool-call events" - let mut tracker = StreamingToolCallTracker::::new() - .with_extract_metadata(|delta| { - delta - .extra - .get("google")? - .get("thought_signature")? - .as_str() - .map(|s| json!({ "thoughtSignature": s })) - }) - .with_build_provider_metadata(|metadata| { - metadata - .as_ref() - .and_then(|m| m.get("thoughtSignature")) - .and_then(|s| s.as_str()) - .map(|s| json!({ "google": { "thoughtSignature": s } })) - }); - - tracker - .process_delta( - &d().index(0) - .id("call_1") - .tool_type("function") - .function_name("fn") - .arguments("{}") - .extra(json!({ "google": { "thought_signature": "sig123" } })), - ) - .unwrap(); - - tracker.flush(); - - let tool_call = tracker - .parts() - .iter() - .find(|p| matches!(p, ToolCallStreamPart::ToolCall { .. })) - .unwrap(); - assert_eq!( - tool_call, - &ToolCallStreamPart::ToolCall { - tool_call_id: "call_1".into(), - tool_name: "fn".into(), - input: "{}".into(), - provider_metadata: Some(json!({ "google": { "thoughtSignature": "sig123" } })), - } - ); -} - -#[test] -fn include_provider_metadata_for_unfinished_tool_calls_finalized_in_flush() { - // TS: "should include provider metadata for unfinished tool calls finalized in flush" - let mut tracker = StreamingToolCallTracker::::new() - .with_extract_metadata(|_| Some(json!({ "custom": { "key": "value" } }))) - .with_build_provider_metadata(|metadata| { - metadata - .as_ref() - .map(|m| json!({ "provider": (*m).clone() })) - }); - - tracker - .process_delta( - &d().index(0) - .id("call_1") - .tool_type("function") - .function_name("fn") - .arguments("{\"incomplete"), - ) - .unwrap(); - - tracker.clear_parts(); - - tracker.flush(); - - let tool_call = tracker - .parts() - .iter() - .find(|p| matches!(p, ToolCallStreamPart::ToolCall { .. })) - .unwrap(); - assert_eq!( - tool_call, - &ToolCallStreamPart::ToolCall { - tool_call_id: "call_1".into(), - tool_name: "fn".into(), - input: "{\"incomplete".into(), - provider_metadata: Some(json!({ "provider": { "custom": { "key": "value" } } })), - } - ); -} - -#[test] -fn not_include_provider_metadata_when_build_returns_none() { - // TS: "should not include providerMetadata when buildToolCallProviderMetadata returns undefined" - let mut tracker = StreamingToolCallTracker::::new() - .with_extract_metadata(|_| None) - .with_build_provider_metadata(|_| None); - - tracker - .process_delta( - &d().index(0) - .id("call_1") - .tool_type("function") - .function_name("fn") - .arguments("{}"), - ) - .unwrap(); - - tracker.flush(); - - let tool_call = tracker - .parts() - .iter() - .find(|p| matches!(p, ToolCallStreamPart::ToolCall { .. })) - .unwrap(); - assert_eq!( - tool_call, - &ToolCallStreamPart::ToolCall { - tool_call_id: "call_1".into(), - tool_name: "fn".into(), - input: "{}".into(), - provider_metadata: None, - } - ); - // Mirrors the TS `not.toHaveProperty('providerMetadata')` assertion: the - // metadata is explicitly `None`, never `Some`. - assert!(matches!( - tool_call, - ToolCallStreamPart::ToolCall { - provider_metadata: None, - .. - } - )); -} - -// == generateId =========================================================== - -#[test] -fn use_custom_generate_id_for_tool_call_ids_when_id_missing_in_fallback() { - // TS: "should use custom generateId for tool call IDs when id is missing in fallback". - // The id is present (`call_1`), so the custom generator is NOT used; the - // original id is kept (mirrors the TS `toolCall.id ?? generateId()` path). - use std::sync::atomic::{AtomicUsize, Ordering}; - static CALLS: AtomicUsize = AtomicUsize::new(0); - - let mut tracker = StreamingToolCallTracker::<()>::new().with_generate_id(|| { - CALLS.fetch_add(1, Ordering::SeqCst); - "custom-id".to_string() - }); - - tracker - .process_delta( - &d().index(0) - .id("call_1") - .tool_type("function") - .function_name("fn") - .arguments("{\"key\": \"val"), - ) - .unwrap(); - - tracker.clear_parts(); - tracker.flush(); - - let tool_call = tracker - .parts() - .iter() - .find(|p| matches!(p, ToolCallStreamPart::ToolCall { .. })) - .unwrap(); - assert_eq!(tool_call_id(tool_call), "call_1"); - // The custom generator must not have been invoked. - assert_eq!(CALLS.load(Ordering::SeqCst), 0); -} - -#[test] -fn use_custom_generate_id_when_id_is_missing() { - // P2-05: when a tool-call delta omits its id, the configured `generateId` - // fallback is invoked (TS `toolCall.id ?? generateId()`). The generated id - // is used for the emitted ToolInputStart and the finalized ToolCall. - use std::sync::atomic::{AtomicUsize, Ordering}; - static CALLS: AtomicUsize = AtomicUsize::new(0); - - let mut tracker = StreamingToolCallTracker::<()>::new().with_generate_id(|| { - CALLS.fetch_add(1, Ordering::SeqCst); - "custom-id".to_string() - }); - - // No `.id(...)` — the generator must fill it in. - tracker - .process_delta( - &d().index(0) - .tool_type("function") - .function_name("fn") - .arguments("{\"key\": \"val"), - ) - .unwrap(); - - // The generator was invoked exactly once for the missing id. - assert_eq!(CALLS.load(Ordering::SeqCst), 1); - - // The emitted start part carries the generated id. - assert_eq!( - tracker.parts(), - &[ - ToolCallStreamPart::ToolInputStart { - id: "custom-id".into(), - tool_name: "fn".into(), - }, - ToolCallStreamPart::ToolInputDelta { - id: "custom-id".into(), - delta: "{\"key\": \"val".into(), - }, - ] - ); - - tracker.clear_parts(); - tracker.flush(); - - let tool_call = tracker - .parts() - .iter() - .find(|p| matches!(p, ToolCallStreamPart::ToolCall { .. })) - .unwrap(); - assert_eq!(tool_call_id(tool_call), "custom-id"); - // No additional generator invocations during flush. - assert_eq!(CALLS.load(Ordering::SeqCst), 1); -} - -// == index bound (P1-10) ================================================== - -#[test] -fn reject_tool_call_index_above_max_index() { - // A remote `index` far above the cap must not resize `tool_calls` to a - // huge vector; it returns `IndexOutOfRange`. - let mut tracker = new_tracker().with_max_index(4); - - // index == max_index is accepted (boundary). - tracker - .process_delta( - &d().index(4) - .id("call_4") - .tool_type("function") - .function_name("fn") - .arguments(""), - ) - .unwrap(); - - // index == max_index + 1 is rejected. - let err = tracker - .process_delta( - &d().index(5) - .id("call_5") - .tool_type("function") - .function_name("fn") - .arguments(""), - ) - .unwrap_err(); - assert_eq!(err, TrackerError::IndexOutOfRange); - assert_eq!(err.to_string(), "Tool call index out of range"); -} - -#[test] -fn default_max_index_rejects_huge_index() { - // The default cap (1024) rejects an absurd index that would otherwise - // resize `tool_calls` to index+1 slots. - let mut tracker = new_tracker(); - let err = tracker - .process_delta( - &d().index(1_000_000) - .id("call_x") - .tool_type("function") - .function_name("fn") - .arguments(""), - ) - .unwrap_err(); - assert_eq!(err, TrackerError::IndexOutOfRange); -} diff --git a/docs/PROJECT-OVERVIEW.md b/docs/PROJECT-OVERVIEW.md index a5442aca..b79cb53a 100644 --- a/docs/PROJECT-OVERVIEW.md +++ b/docs/PROJECT-OVERVIEW.md @@ -165,8 +165,8 @@ aimux/ │ ├── native protocols # standalone model + convert, handles provider-specific differences │ ├── OpenAI compatible # registry-backed: provider-registry.json + provider(name, ...) entry (RFC-0017 phase 4) │ └── modalities/search # voice / image / video / search implementations -├── aimux-stream # SSE / NDJSON streaming parsing -├── aimux-provider-utils # One-exchange HTTP helpers, response handlers, API-key loading +├── aimux-stream # SSE decoding +├── aimux-provider-utils # One-exchange HTTP helpers, response handlers, API-key loading, streamed tool-call tracking ├── aimux-ffi # C ABI (FFI infrastructure, shared by all bindings) └── bindings/ # 6 language bindings ├── node/ # napi-rs v3 + typed TS wrapper