Skip to main content

headless_lms_server/controllers/
mock_azure.rs

1use crate::controllers::mock_document_storage::{MOCK_DOCUMENTS, MockDocument};
2use crate::prelude::*;
3use headless_lms_chatbot::{
4    azure_chatbot::azure::{
5        protocol::{InputItem, LLMRequestResponseFormatParam},
6        tools::AzureLLMToolDefinition,
7    },
8    chart_spec_generation::{
9        DATA_SAMPLE_HEADING, EXISTING_SPEC_HEADING, USER_PROMPT_PREFIX as CHART_SPEC_REQUEST,
10    },
11    chatbot_tools::{
12        ChatbotToolDeclaration,
13        client_tools::ask_multiple_choice_question::AskMultipleChoiceQuestionTool,
14        custom_tools::course_structure::CourseStructureTool, get_chatbot_tool_definitions,
15        tool_is_answered_by_client,
16    },
17    cms_ai_suggestion::RESPONSE_FORMAT_NAME as CMS_SUGGESTION_FORMAT,
18    course_description_summary::RESPONSE_FORMAT_NAME as COURSE_DESCRIPTION_FORMAT,
19    llm_utils::AzureCompletionRequest,
20    message_suggestion::RESPONSE_FORMAT_NAME as MESSAGE_SUGGESTION_FORMAT,
21    prompt_creation::RESPONSE_FORMAT_NAME as PROMPT_CREATION_FORMAT,
22};
23use headless_lms_utils::azure_embedding::{
24    Embedding, EmbeddingRequest, EmbeddingResponse, EmbeddingResponseUsage,
25};
26use serde_json::{Value, json};
27
28/// Anywhere in a chat message, this makes the mock answer with a function call instead of a
29/// text answer, which is how a test drives the chatbot's tool loop. The chatbot then runs the
30/// tool and asks again with the tool output as the last input item, and the mock answers that
31/// with a text round, completing a two-round turn.
32const TOOL_CALL_TRIGGER: &str = "!MOCK_TOOL_CALL!";
33
34/// Like [TOOL_CALL_TRIGGER], but the call is one only the client can answer, so the chatbot
35/// suspends the turn instead of running anything. The turn is finished by the tool-response
36/// endpoint, whose request ends in the tool output and so gets the same text round.
37const CLIENT_TOOL_CALL_TRIGGER: &str = "!MOCK_CLIENT_TOOL_CALL!";
38
39/// Opens the trigger `!MOCK_TOOL_CALL:<tool_name>:<arguments>!`, which makes the mock call any
40/// registered tool, so exercising a new one end to end needs no scenario of its own here.
41const TOOL_CALL_BY_NAME_PREFIX: &str = "!MOCK_TOOL_CALL:";
42
43/// Drives an answer Azure cut short by `max_output_tokens`: the round ends on
44/// `response.incomplete` instead of `response.completed`.
45const INCOMPLETE_ANSWER_TRIGGER: &str = "!MOCK_INCOMPLETE_ANSWER!";
46
47/// Drives a stream a proxy closed cleanly before Azure ever sent a terminal event — neither
48/// `response.completed` nor `response.incomplete`.
49const TRUNCATED_STREAM_TRIGGER: &str = "!MOCK_TRUNCATED_STREAM!";
50
51/// Drives a search round whose completed output fails to parse as the chatbot's search-output
52/// schema: the same plain-text failure shape a real search backend can return, but on a
53/// `completed` item rather than the in-progress placeholder every other scenario here uses.
54const MALFORMED_SEARCH_OUTPUT_TRIGGER: &str = "!MOCK_MALFORMED_SEARCH_OUTPUT!";
55
56/// Drives a round that includes an output item kind the chatbot has no variant for.
57const UNKNOWN_ITEM_TYPE_TRIGGER: &str = "!MOCK_UNKNOWN_ITEM_TYPE!";
58
59/// The trigger text that makes the mock call `tool_name` with `arguments`.
60fn tool_call_by_name_trigger(tool_name: &str, arguments: &str) -> String {
61    format!("{TOOL_CALL_BY_NAME_PREFIX}{tool_name}:{arguments}!")
62}
63
64/// A tool call a message asked the mock to make.
65struct TriggeredToolCall {
66    tool_name: String,
67    arguments: String,
68}
69
70/// Reads a [TOOL_CALL_BY_NAME_PREFIX] trigger out of `message`.
71///
72/// A learner can type the trigger into the chat, so a name no registry claims has to read as an
73/// ordinary message rather than reach the chatbot as a hallucinated call. `!` closes the trigger
74/// and so cannot appear in `arguments`.
75fn parse_tool_call_by_name(message: &str) -> Option<TriggeredToolCall> {
76    let (_, rest) = message.split_once(TOOL_CALL_BY_NAME_PREFIX)?;
77    let (call, _) = rest.split_once('!')?;
78    let (tool_name, arguments) = call.split_once(':')?;
79    is_registered_tool(tool_name).then(|| TriggeredToolCall {
80        tool_name: tool_name.to_string(),
81        arguments: arguments.to_string(),
82    })
83}
84
85/// Whether either registry claims `tool_name`.
86fn is_registered_tool(tool_name: &str) -> bool {
87    tool_is_answered_by_client(tool_name)
88        || get_chatbot_tool_definitions()
89            .iter()
90            .any(|definition| match definition {
91                AzureLLMToolDefinition::Function(function) => function.name == tool_name,
92                AzureLLMToolDefinition::Search(_) => false,
93            })
94}
95
96/// The parts of a request the mock picks its answer from.
97struct MockRequest {
98    /// The text of the last input message, or `None` when the request ends in a tool output
99    /// instead. The other two endings never reach here.
100    message: Option<String>,
101    /// Every message in the request, for recognising a round of a conversation whose latest
102    /// message says nothing about which feature it belongs to.
103    conversation: Vec<String>,
104    /// The structured output schema the answer has to parse as, `None` when none was named.
105    format_name: Option<String>,
106    /// Whether the caller asked for any valid JSON object instead of a named schema.
107    wants_json_object: bool,
108    /// Whether the caller reads the answer as a Server-Sent Events stream or as one JSON object.
109    stream: bool,
110}
111
112impl MockRequest {
113    /// A streamed chat request carrying `message`.
114    fn chat(message: &str) -> Self {
115        MockRequest {
116            message: Some(message.to_string()),
117            conversation: vec![message.to_string()],
118            format_name: None,
119            wants_json_object: false,
120            stream: true,
121        }
122    }
123
124    /// A streamed request resuming a tool loop, which ends in the tool's output rather than in a
125    /// message.
126    fn after_tool_run() -> Self {
127        MockRequest {
128            message: None,
129            conversation: Vec::new(),
130            format_name: None,
131            wants_json_object: false,
132            stream: true,
133        }
134    }
135
136    /// A request for structured output in `format_name`. Its message is one that would drive a
137    /// function call round if the message decided anything here, which it must not.
138    fn structured_output(format_name: &str) -> Self {
139        MockRequest {
140            message: Some(TOOL_CALL_TRIGGER.to_string()),
141            conversation: vec![TOOL_CALL_TRIGGER.to_string()],
142            format_name: Some(format_name.to_string()),
143            wants_json_object: false,
144            stream: false,
145        }
146    }
147
148    /// A request for any valid JSON object, whose conversation contains `request_text`.
149    fn json_object(request_text: &str) -> Self {
150        MockRequest {
151            message: Some(request_text.to_string()),
152            conversation: vec![request_text.to_string()],
153            format_name: None,
154            wants_json_object: true,
155            stream: false,
156        }
157    }
158
159    /// Whether this asks for the structured output named `format_name`. False for a streamed
160    /// request even when it names the schema, since the answer to one is a whole JSON object that
161    /// a streaming caller cannot read.
162    fn wants_format(&self, format_name: &str) -> bool {
163        !self.stream && self.format_name.as_deref() == Some(format_name)
164    }
165
166    /// Whether this asks for a JSON object and is part of a conversation that opened with
167    /// `request_text`. A JSON-mode request names no schema, so the feature it belongs to can only
168    /// be told from its prompt -- and not from the latest message alone, which on a repair round is
169    /// the correction rather than the request.
170    fn wants_json_object_for(&self, request_text: &str) -> bool {
171        !self.stream
172            && self.wants_json_object
173            && self
174                .conversation
175                .iter()
176                .any(|message| message.contains(request_text))
177    }
178
179    /// The part of the prompt that `heading` introduces, ending where `next_heading` begins or at
180    /// the end of the message. None when no message in the conversation carries `heading`.
181    ///
182    /// Reads the first message that has it: on a repair round the latest message is the correction
183    /// rather than the request the headings belong to.
184    fn prompt_section(&self, heading: &str, next_heading: &str) -> Option<&str> {
185        let (_, after) = self
186            .conversation
187            .iter()
188            .find_map(|message| message.split_once(heading))?;
189        Some(after.split(next_heading).next().unwrap_or(after))
190    }
191
192    /// The tool call a streamed chat message asks the mock to make, if any.
193    fn triggered_tool_call(&self) -> Option<TriggeredToolCall> {
194        if !self.stream {
195            return None;
196        }
197        parse_tool_call_by_name(self.message.as_deref()?)
198    }
199
200    /// Whether this is a streamed chat message containing `trigger`.
201    fn message_contains(&self, trigger: &str) -> bool {
202        self.stream
203            && self
204                .message
205                .as_deref()
206                .is_some_and(|message| message.contains(trigger))
207    }
208}
209
210/// One shape of request the mock answers, and the answer it gives.
211///
212/// The tests drive every registered scenario through its own [`example`](Scenario::example), so
213/// registering a round here is what gets it verified at all.
214struct Scenario {
215    /// Names the scenario in the handler's log line and in test failures.
216    name: &'static str,
217    matches: fn(&MockRequest) -> bool,
218    /// Builds the answer to `request` against the base url its document urls have to point at.
219    respond: fn(&MockRequest, &str) -> String,
220    /// A request this scenario answers.
221    #[cfg_attr(not(test), allow(dead_code))]
222    example: fn() -> MockRequest,
223}
224
225/// The first scenario that matches answers, so the default chat answer, which takes any streamed
226/// message at all, comes last.
227///
228/// A blocking caller parses the whole body as one JSON object and a streaming one reads it event by
229/// event, so no answer suits both, and every scenario says which kind it is. Among the blocking
230/// ones the schema alone decides: a learner is free to type anything into the chat, so no part of a
231/// message may reach a feature's answer.
232const SCENARIOS: &[Scenario] = &[
233    Scenario {
234        name: "the next message suggestion",
235        matches: |request| request.wants_format(MESSAGE_SUGGESTION_FORMAT),
236        respond: |_, _| blocking_response(MESSAGE_SUGGESTION_PAYLOAD),
237        example: || MockRequest::structured_output(MESSAGE_SUGGESTION_FORMAT),
238    },
239    Scenario {
240        name: "the CMS paragraph suggestion",
241        matches: |request| request.wants_format(CMS_SUGGESTION_FORMAT),
242        respond: |_, _| blocking_response(CMS_SUGGESTION_PAYLOAD),
243        example: || MockRequest::structured_output(CMS_SUGGESTION_FORMAT),
244    },
245    Scenario {
246        name: "the course description summary",
247        matches: |request| request.wants_format(COURSE_DESCRIPTION_FORMAT),
248        respond: |_, _| blocking_response(COURSE_DESCRIPTION_PAYLOAD),
249        example: || MockRequest::structured_output(COURSE_DESCRIPTION_FORMAT),
250    },
251    Scenario {
252        name: "the chart block specification",
253        matches: |request| request.wants_json_object_for(CHART_SPEC_REQUEST),
254        respond: |request, _| blocking_response(&chart_spec_payload(request)),
255        example: || MockRequest::json_object(CHART_SPEC_REQUEST),
256    },
257    Scenario {
258        name: "the client tool call round",
259        matches: |request| request.message_contains(CLIENT_TOOL_CALL_TRIGGER),
260        respond: |_, _| {
261            function_call_round(
262                <AskMultipleChoiceQuestionTool as ChatbotToolDeclaration>::NAME,
263                MOCK_MULTIPLE_CHOICE_ARGUMENTS,
264            )
265        },
266        example: || MockRequest::chat(CLIENT_TOOL_CALL_TRIGGER),
267    },
268    Scenario {
269        name: "the tool call round for a named tool",
270        matches: |request| request.triggered_tool_call().is_some(),
271        respond: |request, _| {
272            let call = request
273                .triggered_tool_call()
274                .expect("the scenario only answers a request carrying a tool call trigger");
275            function_call_round(&call.tool_name, &call.arguments)
276        },
277        example: || {
278            MockRequest::chat(&tool_call_by_name_trigger(
279                <CourseStructureTool as ChatbotToolDeclaration>::NAME,
280                "{}",
281            ))
282        },
283    },
284    Scenario {
285        name: "the function call round",
286        matches: |request| request.message_contains(TOOL_CALL_TRIGGER),
287        respond: |_, _| {
288            function_call_round(<CourseStructureTool as ChatbotToolDeclaration>::NAME, "{}")
289        },
290        example: || MockRequest::chat(TOOL_CALL_TRIGGER),
291    },
292    Scenario {
293        name: "the answer after a tool ran",
294        matches: |request| request.stream && request.message.is_none(),
295        respond: |_, _| tool_answer_round(),
296        example: MockRequest::after_tool_run,
297    },
298    Scenario {
299        name: "the prompt and first message generation",
300        matches: |request| request.wants_format(PROMPT_CREATION_FORMAT),
301        respond: |_, _| blocking_response(PROMPT_CREATION_PAYLOAD),
302        example: || MockRequest::structured_output(PROMPT_CREATION_FORMAT),
303    },
304    Scenario {
305        name: "an answer cut short by the token limit",
306        matches: |request| request.message_contains(INCOMPLETE_ANSWER_TRIGGER),
307        respond: |_, _| incomplete_answer_round(),
308        example: || MockRequest::chat(INCOMPLETE_ANSWER_TRIGGER),
309    },
310    Scenario {
311        name: "a stream that ends before response.completed",
312        matches: |request| request.message_contains(TRUNCATED_STREAM_TRIGGER),
313        respond: |_, _| truncated_stream_round(),
314        example: || MockRequest::chat(TRUNCATED_STREAM_TRIGGER),
315    },
316    Scenario {
317        name: "a completed search output that fails to parse",
318        matches: |request| request.message_contains(MALFORMED_SEARCH_OUTPUT_TRIGGER),
319        respond: |_, _| malformed_search_output_round(),
320        example: || MockRequest::chat(MALFORMED_SEARCH_OUTPUT_TRIGGER),
321    },
322    Scenario {
323        name: "an output item type the chatbot does not know",
324        matches: |request| request.message_contains(UNKNOWN_ITEM_TYPE_TRIGGER),
325        respond: |_, _| unknown_item_type_round(),
326        example: || MockRequest::chat(UNKNOWN_ITEM_TYPE_TRIGGER),
327    },
328    Scenario {
329        name: "the default chat answer",
330        matches: |request| request.stream && request.message.is_some(),
331        respond: |_, base_url| search_and_text_round(base_url),
332        example: || MockRequest::chat("Tell me more"),
333    },
334];
335
336/// The scenario that answers `request`, or `None` when the mock answers nothing like it.
337fn pick_scenario(request: &MockRequest) -> Option<&'static Scenario> {
338    SCENARIOS
339        .iter()
340        .find(|scenario| (scenario.matches)(request))
341}
342
343/// GET /api/v0/mock-azure/api/projects/test/openai/v1/responses
344/// POST /api/v0/mock-azure/api/projects/test/openai/v1/responses
345///
346/// Stands in for the Azure Responses API while the chatbot runs in test mode. Answers with the
347/// first scenario in [`SCENARIOS`] whose request shape matches, and 400s on a request no scenario
348/// answers.
349async fn mock_azure_chat_responses(
350    app_conf: web::Data<ApplicationConfiguration>,
351    payload: web::Json<AzureCompletionRequest>,
352) -> ControllerResult<String> {
353    assert!(app_conf.test_chatbot && app_conf.test_mode);
354
355    let last_input_item = &payload
356        .base
357        .input
358        .last()
359        .ok_or_else(|| {
360            controller_err!(
361                BadRequest,
362                "No messages in request, there should be at least one."
363            )
364        })?
365        .message_type;
366
367    let message = match last_input_item {
368        InputItem::Message { content, .. } => Some(content.clone().get_content_text()),
369        InputItem::FunctionCallOutput { .. } => None,
370        InputItem::FunctionCall { .. } | InputItem::Reasoning { .. } => {
371            return Err(controller_err!(
372                BadRequest,
373                "The mock has no response for a request that ends in a function call or a reasoning item."
374            ));
375        }
376    };
377
378    let requested_format = payload
379        .base
380        .text
381        .as_ref()
382        .and_then(|text| text.format.as_ref());
383    let request = MockRequest {
384        message,
385        conversation: payload
386            .base
387            .input
388            .iter()
389            .filter_map(|item| match &item.message_type {
390                InputItem::Message { content, .. } => Some(content.clone().get_content_text()),
391                _ => None,
392            })
393            .collect(),
394        format_name: match requested_format {
395            Some(LLMRequestResponseFormatParam::JsonSchema { name, .. }) => Some(name.clone()),
396            _ => None,
397        },
398        wants_json_object: matches!(
399            requested_format,
400            Some(LLMRequestResponseFormatParam::JsonObject)
401        ),
402        stream: payload.stream,
403    };
404    let scenario = pick_scenario(&request).ok_or_else(|| {
405        controller_err!(
406            BadRequest,
407            "The mock has no response for this shape of request."
408        )
409    })?;
410    debug!(scenario = scenario.name, "Answering as the mock Azure API");
411    let res = (scenario.respond)(&request, &app_conf.base_url);
412
413    let token = skip_authorize();
414    token.authorized_ok(res)
415}
416
417/// Renders `events` as a Server-Sent Events body in the order given, stamping every `data:` object
418/// with its own event name as `type`, the way Azure repeats it. The chatbot reads the `event:` line
419/// to decide what the `data:` line after it means, so the pairing and the order are what make a
420/// round parse as a tool call or as text.
421fn sse_body(events: Vec<(&str, Value)>) -> String {
422    events
423        .into_iter()
424        .map(|(event, mut data)| {
425            if let Some(object) = data.as_object_mut() {
426                object.insert("type".to_string(), json!(event));
427            }
428            format!("event: {event}\ndata: {data}\n\n")
429        })
430        .collect()
431}
432
433/// Wraps a round's own `events` in the lifecycle events every round shares: `response.created`,
434/// which is where the chatbot picks up the response id it needs before the round is classified,
435/// and the terminal `response.completed`.
436fn round(response_id: &str, events: Vec<(&'static str, Value)>, usage: Value) -> String {
437    let mut all = vec![(
438        "response.created",
439        json!({"response": response_object(response_id)}),
440    )];
441    all.extend(events);
442    all.push((
443        "response.completed",
444        json!({"response": completed_response_object(response_id, usage)}),
445    ));
446    sse_body(all)
447}
448
449/// What one round was billed for. `cached_tokens` is the part of the input Azure served from the
450/// prompt cache, and the rest of it is what the round wrote there.
451fn usage(
452    input_tokens: u32,
453    cached_tokens: u32,
454    output_tokens: u32,
455    reasoning_tokens: u32,
456) -> Value {
457    json!({
458        "input_tokens": input_tokens,
459        "input_tokens_details": {
460            "cached_tokens": cached_tokens,
461            "cache_write_tokens": input_tokens.saturating_sub(cached_tokens),
462        },
463        "output_tokens": output_tokens,
464        "output_tokens_details": {"reasoning_tokens": reasoning_tokens},
465        "total_tokens": input_tokens + output_tokens,
466    })
467}
468
469/// The response object of a lifecycle event before the last one. Only `id` and a possible `error`
470/// are read by the chatbot; Azure sends the full request parameters here as well. See
471/// [`completed_response_object`] for the terminal event.
472fn response_object(response_id: &str) -> Value {
473    json!({
474        "id": response_id,
475        "object": "response",
476        "status": "in_progress",
477        "usage": null,
478    })
479}
480
481/// The response object of the `response.completed` event, the only lifecycle event Azure reports
482/// token usage and the reasoning context on.
483fn completed_response_object(response_id: &str, usage: Value) -> Value {
484    json!({
485        "id": response_id,
486        "object": "response",
487        "status": "completed",
488        "reasoning": {"effort": "medium", "summary": null, "context": "current_turn"},
489        "usage": usage,
490    })
491}
492
493/// One assistant message as an output item, in the two shapes a round streams it in: `content` is
494/// empty while the item is in progress and carries the whole text once it is done.
495fn message_item(item_id: &str, response_id: &str, content: Value, status: &str) -> Value {
496    json!({
497        "type": "message",
498        "id": item_id,
499        "response_id": response_id,
500        "phase": "final_answer",
501        "role": "assistant",
502        "content": content,
503        "status": status,
504    })
505}
506
507/// The events that stream one assistant message: the item, its content part, one event per delta,
508/// and the finished item. `output_index` is the message's place among the round's output items, so
509/// it depends on how many items the round emitted before this one.
510fn message_item_events(
511    item_id: &str,
512    response_id: &str,
513    output_index: u32,
514    deltas: &[&str],
515) -> Vec<(&'static str, Value)> {
516    let text = deltas.concat();
517
518    let mut events = vec![
519        (
520            "response.output_item.added",
521            json!({
522                "output_index": output_index,
523                "item": message_item(item_id, response_id, json!([]), "in_progress"),
524            }),
525        ),
526        (
527            "response.content_part.added",
528            json!({
529                "content_index": 0,
530                "item_id": item_id,
531                "output_index": output_index,
532                "part": { "type": "output_text", "text": "" },
533            }),
534        ),
535    ];
536    events.extend(deltas.iter().map(|delta| {
537        (
538            "response.output_text.delta",
539            json!({
540                "content_index": 0,
541                "item_id": item_id,
542                "output_index": output_index,
543                "delta": delta,
544            }),
545        )
546    }));
547    events.extend([
548        (
549            "response.output_text.done",
550            json!({
551                "content_index": 0,
552                "item_id": item_id,
553                "output_index": output_index,
554                "text": text,
555            }),
556        ),
557        (
558            "response.content_part.done",
559            json!({
560                "content_index": 0,
561                "item_id": item_id,
562                "output_index": output_index,
563                "part": { "type": "output_text", "text": text },
564            }),
565        ),
566        (
567            "response.output_item.done",
568            json!({
569                "output_index": output_index,
570                "item": message_item(
571                    item_id,
572                    response_id,
573                    json!([{ "type": "output_text", "text": text }]),
574                    "completed",
575                ),
576            }),
577        ),
578    ]);
579    events
580}
581
582/// The events that stream the reasoning item a round emits before it calls anything or answers,
583/// which is what makes it the round's first output item.
584fn reasoning_item_events(response_id: &str) -> Vec<(&'static str, Value)> {
585    let item = reasoning_item(response_id);
586    vec![
587        (
588            "response.output_item.added",
589            json!({"output_index": 0, "item": item}),
590        ),
591        (
592            "response.output_item.done",
593            json!({"output_index": 0, "item": item}),
594        ),
595    ]
596}
597
598/// The item [`reasoning_item_events`] streams, unchanged in both of its events.
599fn reasoning_item(response_id: &str) -> Value {
600    json!({
601        "type": "reasoning",
602        "id": format!("rs_{}", Uuid::new_v4()),
603        "response_id": response_id,
604        "summary": [],
605        // Azure returns this whenever `store` is false, and the chatbot only replays a reasoning
606        // item that has it, so without it the mock never exercises the replay path.
607        "encrypted_content": "mock-encrypted-reasoning",
608    })
609}
610
611/// The question [CLIENT_TOOL_CALL_TRIGGER] makes the mock ask. Has to pass the tool's own
612/// argument validation, which the chatbot runs before it suspends the turn.
613const MOCK_MULTIPLE_CHOICE_ARGUMENTS: &str =
614    r#"{"question":"Which loop do you mean?","choices":["while","for"]}"#;
615
616/// A round in which the model calls `tool_name` with `arguments`.
617///
618/// Every tool it is used with works without a user, so the round works for an anonymous course
619/// material visitor.
620///
621/// The chatbot classifies the round from the function call item it announces and passes the tool
622/// parser only what follows that item, so the completed call has to arrive in a later
623/// `output_item.done`, and the round must contain no text delta at all.
624fn function_call_round(tool_name: &str, arguments: &str) -> String {
625    let response_id = format!("resp_{}", Uuid::new_v4());
626    let item_id = format!("fc_{}", Uuid::new_v4());
627    let call_id = format!("call_{}", Uuid::new_v4());
628
629    let function_call = |arguments: &str, status: &str| {
630        json!({
631            "type": "function_call",
632            "id": item_id,
633            "response_id": response_id,
634            "call_id": call_id,
635            "name": tool_name,
636            "arguments": arguments,
637            "status": status,
638        })
639    };
640    let mut events = reasoning_item_events(&response_id);
641    events.extend([
642        (
643            "response.output_item.added",
644            json!({"output_index": 1, "item": function_call("", "in_progress")}),
645        ),
646        (
647            "response.function_call_arguments.delta",
648            json!({"item_id": item_id, "output_index": 1, "delta": arguments}),
649        ),
650        (
651            "response.function_call_arguments.done",
652            json!({"item_id": item_id, "output_index": 1, "arguments": arguments}),
653        ),
654        (
655            "response.output_item.done",
656            json!({"output_index": 1, "item": function_call(arguments, "completed")}),
657        ),
658    ]);
659
660    round(&response_id, events, usage(42, 0, 88, 64))
661}
662
663/// The text answer the model gives once a tool has run.
664fn tool_answer_round() -> String {
665    let response_id = format!("resp_{}", Uuid::new_v4());
666    let item_id = format!("msg_{}", Uuid::new_v4());
667    let deltas = [
668        "Here", " is", " the", " mock", " answer", " after", " a", " tool", " ran.",
669    ];
670
671    round(
672        &response_id,
673        message_item_events(&item_id, &response_id, 0, &deltas),
674        // Second round of the same turn, so what the first round wrote to the cache comes back as
675        // a cache read here.
676        usage(96, 42, 24, 8),
677    )
678}
679
680/// An answer `max_output_tokens` cut short: it ends on `response.incomplete`, never on
681/// `response.completed`, with only a partial delta having streamed.
682fn incomplete_answer_round() -> String {
683    let response_id = format!("resp_{}", Uuid::new_v4());
684    let item_id = format!("msg_{}", Uuid::new_v4());
685
686    let events = vec![
687        (
688            "response.created",
689            json!({"response": response_object(&response_id)}),
690        ),
691        (
692            "response.output_item.added",
693            json!({
694                "output_index": 0,
695                "item": message_item(&item_id, &response_id, json!([]), "in_progress"),
696            }),
697        ),
698        (
699            "response.output_text.delta",
700            json!({
701                "content_index": 0,
702                "item_id": item_id,
703                "output_index": 0,
704                "delta": "This answer gets cut",
705            }),
706        ),
707        (
708            "response.incomplete",
709            json!({"response": {
710                "id": response_id,
711                "object": "response",
712                "status": "incomplete",
713                "incomplete_details": {"reason": "max_output_tokens"},
714                "usage": usage(30, 0, 8, 0),
715            }}),
716        ),
717    ];
718    sse_body(events)
719}
720
721/// A stream a proxy closed cleanly before Azure ever sent `response.completed` or
722/// `response.incomplete` — the shape a clean EOF produces, as opposed to the idle-timeout error a
723/// stalled connection produces.
724fn truncated_stream_round() -> String {
725    let response_id = format!("resp_{}", Uuid::new_v4());
726    let item_id = format!("fc_{}", Uuid::new_v4());
727    let call_id = format!("call_{}", Uuid::new_v4());
728
729    sse_body(vec![
730        (
731            "response.created",
732            json!({"response": response_object(&response_id)}),
733        ),
734        (
735            "response.output_item.done",
736            json!({
737                "output_index": 0,
738                "item": {
739                    "type": "function_call",
740                    "id": item_id,
741                    "response_id": response_id,
742                    "call_id": call_id,
743                    "name": <CourseStructureTool as ChatbotToolDeclaration>::NAME,
744                    "arguments": "{}",
745                    "status": "completed",
746                },
747            }),
748        ),
749    ])
750}
751
752/// A completed Azure AI Search output whose `output` is the same plain-text failure shape a real
753/// search backend can return, rather than the `AISearchOutput` JSON the round otherwise expects —
754/// on a `completed` item, unlike the in-progress placeholder [`search_and_text_round`] always
755/// resolves through.
756fn malformed_search_output_round() -> String {
757    let response_id = format!("resp_{}", Uuid::new_v4());
758    let call_id = format!("call_{}", Uuid::new_v4());
759    let search_item_id = format!("fc_{}", Uuid::new_v4());
760    let output_item_id = format!("fco_{}", Uuid::new_v4());
761    let message_item_id = format!("msg_{}", Uuid::new_v4());
762
763    let mut events = reasoning_item_events(&response_id);
764    events.extend([
765        (
766            "response.output_item.done",
767            json!({
768                "output_index": 1,
769                "item": {
770                    "type": "azure_ai_search_call",
771                    "id": search_item_id,
772                    "response_id": response_id,
773                    "call_id": call_id,
774                    "arguments": r#"{"query":"tell me more"}"#,
775                    "status": "completed",
776                },
777            }),
778        ),
779        (
780            "response.output_item.done",
781            json!({
782                "output_index": 2,
783                "item": {
784                    "type": "azure_ai_search_call_output",
785                    "id": output_item_id,
786                    "response_id": response_id,
787                    "call_id": call_id,
788                    "output": "remote tool call failed",
789                    "status": "completed",
790                },
791            }),
792        ),
793    ]);
794    events.extend(message_item_events(
795        &message_item_id,
796        &response_id,
797        3,
798        &["Sorry", ", search is unavailable."],
799    ));
800
801    round(&response_id, events, usage(38, 0, 40, 32))
802}
803
804/// A round that includes an output item kind the chatbot has no variant for, between the
805/// reasoning item and the answer, the way an Azure feature this code predates would arrive.
806fn unknown_item_type_round() -> String {
807    let response_id = format!("resp_{}", Uuid::new_v4());
808    let item_id = format!("ws_{}", Uuid::new_v4());
809    let message_item_id = format!("msg_{}", Uuid::new_v4());
810
811    let mut events = reasoning_item_events(&response_id);
812    events.push((
813        "response.output_item.done",
814        json!({
815            "output_index": 1,
816            "item": {
817                "type": "web_search_call",
818                "id": item_id,
819                "response_id": response_id,
820                "status": "completed",
821            },
822        }),
823    ));
824    events.extend(message_item_events(
825        &message_item_id,
826        &response_id,
827        2,
828        &["Handled", " gracefully."],
829    ));
830
831    round(&response_id, events, usage(20, 0, 20, 16))
832}
833
834/// The default chat answer: a search of the course material, the results it returns, and a text
835/// answer citing them.
836fn search_and_text_round(base_url: &str) -> String {
837    let response_id = format!("resp_{}", Uuid::new_v4());
838    let call_id = format!("call_{}", Uuid::new_v4());
839    let search_item_id = format!("fc_{}", Uuid::new_v4());
840    let output_item_id = format!("fco_{}", Uuid::new_v4());
841    let message_item_id = format!("msg_{}", Uuid::new_v4());
842
843    let search_call = |arguments: &str, status: &str| {
844        json!({
845            "type": "azure_ai_search_call",
846            "id": search_item_id,
847            "response_id": response_id,
848            "call_id": call_id,
849            "arguments": arguments,
850            "status": status,
851        })
852    };
853    let search_call_output = |output: &str, status: &str| {
854        json!({
855            "type": "azure_ai_search_call_output",
856            "id": output_item_id,
857            "response_id": response_id,
858            "call_id": call_id,
859            "output": output,
860            "status": status,
861        })
862    };
863
864    let mut events = reasoning_item_events(&response_id);
865    events.extend([
866        (
867            "response.output_item.added",
868            json!({"output_index": 1, "item": search_call("", "in_progress")}),
869        ),
870        (
871            "response.output_item.done",
872            json!({
873                "output_index": 1,
874                "item": search_call(r#"{"query":"tell me more"}"#, "completed"),
875            }),
876        ),
877        (
878            "response.output_item.added",
879            json!({"output_index": 2, "item": search_call_output("[]", "in_progress")}),
880        ),
881        (
882            "response.output_item.done",
883            json!({
884                "output_index": 2,
885                "item": search_call_output(&search_results(base_url), "completed"),
886            }),
887        ),
888    ]);
889    events.extend(message_item_events(
890        &message_item_id,
891        &response_id,
892        3,
893        &SEARCH_ANSWER_DELTAS,
894    ));
895
896    round(&response_id, events, usage(38, 0, 79, 64))
897}
898
899/// The default round's answer, one delta per element. Each `【x:y†source】` is a citation marker the
900/// frontend replaces with a link to the document it points at.
901const SEARCH_ANSWER_DELTAS: [&str; 12] = [
902    "Hello",
903    "!",
904    " How",
905    " can",
906    " I",
907    " assist",
908    " 【0:2†source】",
909    " you",
910    " 【0:1†source】",
911    " today",
912    "?",
913    "【0:2†source】",
914];
915
916/// What the search returns, as the JSON string Azure nests it in. Only `get_urls` is read: the
917/// chatbot fetches each of those to build the answer's citations, so both they and the hits come
918/// from [`MOCK_DOCUMENTS`], the documents the mock document storage serves.
919fn search_results(base_url: &str) -> String {
920    let [document1, document2, document3] = &MOCK_DOCUMENTS;
921    let hit = |id: &str, document: &MockDocument, content: &str| {
922        json!({
923            "id": id,
924            "content": content,
925            "filepath": document.id,
926            "title": document.title,
927            "url": "",
928            "score": 0.016666668,
929            "knowledgeSourceIndex": 0,
930        })
931    };
932    let get_urls: Vec<String> = MOCK_DOCUMENTS
933        .iter()
934        .map(|document| {
935            format!(
936                "{base_url}/api/v0/mock-document-storage/test/documents/{}",
937                document.id
938            )
939        })
940        .collect();
941
942    json!({
943        "documents": [
944            hit(
945                "doc1",
946                document1,
947                "This chunk is a snippet from page {} of the course {}. Mock test page content This is test content blah",
948            ),
949            hit(
950                "doc2",
951                document2,
952                "Mock test page content 2 This is another test page.",
953            ),
954            // A second hit on doc1's page, so it repeats that page's title and filepath and only
955            // the chunk is document3's.
956            hit("doc3", document1, document3.chunk),
957        ],
958        "get_urls": get_urls,
959    })
960    .to_string()
961}
962
963/// A response to a request that asked for structured output instead of a stream: one whole JSON
964/// object, whose text content is `payload`, the JSON the caller's own schema describes.
965fn blocking_response(payload: &str) -> String {
966    let response_id = format!("resp_{}", Uuid::new_v4());
967    let item_id = format!("msg_{}", Uuid::new_v4());
968
969    let mut response = completed_response_object(&response_id, usage(30, 0, 15, 0));
970    response["output"] = json!([message_item(
971        &item_id,
972        &response_id,
973        json!([{ "type": "output_text", "text": payload }]),
974        "completed",
975    )]);
976    response.to_string()
977}
978
979/// The suggestions the chatbot offers as the learner's next message.
980const MESSAGE_SUGGESTION_PAYLOAD: &str =
981    r#"{"suggestions":["Can you pls help me?","Nice weather we're having.","Hello?"]}"#;
982
983/// The rewrites the CMS offers for a paragraph.
984const CMS_SUGGESTION_PAYLOAD: &str = r#"{"suggestions":["Mock suggestion 1: The paragraph has been improved.","Mock suggestion 2: Here is an alternative version of the paragraph.","Mock suggestion 3: A third distinct rewrite of the paragraph."]}"#;
985
986/// A field of the teacher's data file, as the mock reads it out of the sample in the prompt.
987struct SampledField {
988    name: String,
989    /// The Vega-Lite type the sampled value calls for.
990    field_type: &'static str,
991}
992
993/// What the chart encodes when the request carries no sample to read fields out of.
994fn placeholder_fields() -> (SampledField, SampledField) {
995    (
996        SampledField {
997            name: "category".to_string(),
998            field_type: "nominal",
999        },
1000        SampledField {
1001            name: "value".to_string(),
1002            field_type: "quantitative",
1003        },
1004    )
1005}
1006
1007/// Splits on the commas that separate an object's members or a CSV line's cells, which is every
1008/// comma outside a quoted string.
1009fn split_unquoted_commas(text: &str) -> Vec<&str> {
1010    let mut cells = Vec::new();
1011    let mut start = 0;
1012    let mut in_string = false;
1013    let mut escaped = false;
1014    for (i, character) in text.char_indices() {
1015        match character {
1016            _ if escaped => escaped = false,
1017            '\\' if in_string => escaped = true,
1018            '"' => in_string = !in_string,
1019            ',' if !in_string => {
1020                cells.push(&text[start..i]);
1021                start = i + character.len_utf8();
1022            }
1023            _ => {}
1024        }
1025    }
1026    cells.push(&text[start..]);
1027    cells
1028}
1029
1030fn unquote(cell: &str) -> &str {
1031    cell.trim().trim_matches('"').trim()
1032}
1033
1034/// The fields of the first row of a JSON array sample, in the order they are written there.
1035///
1036/// Read out of the text rather than through serde_json, whose maps sort an object's keys and would
1037/// so hand the chart its axes in alphabetical order. A quoted value is a string however numeric it
1038/// looks, which is what tells a year apart from a label.
1039fn json_sample_fields(sample: &str) -> Vec<SampledField> {
1040    let Some(open) = sample.find('{') else {
1041        return Vec::new();
1042    };
1043    let Some(close) = sample[open..].find('}') else {
1044        return Vec::new();
1045    };
1046    split_unquoted_commas(&sample[open + 1..open + close])
1047        .into_iter()
1048        .filter_map(|member| {
1049            let (key, value) = member.split_once(':')?;
1050            let name = unquote(key).to_string();
1051            let numeric = !value.trim().starts_with('"') && value.trim().parse::<f64>().is_ok();
1052            (!name.is_empty()).then_some(SampledField {
1053                name,
1054                field_type: if numeric { "quantitative" } else { "nominal" },
1055            })
1056        })
1057        .collect()
1058}
1059
1060/// The columns of a CSV sample, typed from the first row of values under the header.
1061fn csv_sample_fields(sample: &str) -> Vec<SampledField> {
1062    let mut rows = sample.lines().filter(|line| !line.trim().is_empty());
1063    let header = rows.next().unwrap_or_default();
1064    let first_row = split_unquoted_commas(rows.next().unwrap_or_default());
1065    split_unquoted_commas(header)
1066        .into_iter()
1067        .enumerate()
1068        .filter_map(|(column, name)| {
1069            let name = unquote(name).to_string();
1070            let numeric = first_row
1071                .get(column)
1072                .is_some_and(|value| unquote(value).parse::<f64>().is_ok());
1073            (!name.is_empty()).then_some(SampledField {
1074                name,
1075                field_type: if numeric { "quantitative" } else { "nominal" },
1076            })
1077        })
1078        .collect()
1079}
1080
1081/// The first two fields of the teacher's data file, or [`placeholder_fields`] when the request
1082/// carries no sample or the sample has fewer than two fields to chart.
1083///
1084/// A real model is told to encode the fields the sample shows it, and a chart of fields the file
1085/// does not have draws nothing at all.
1086fn charted_fields(request: &MockRequest) -> (SampledField, SampledField) {
1087    let Some(sample) = request.prompt_section(DATA_SAMPLE_HEADING, EXISTING_SPEC_HEADING) else {
1088        return placeholder_fields();
1089    };
1090    let fields = if sample.trim_start().starts_with(['[', '{']) {
1091        json_sample_fields(sample)
1092    } else {
1093        csv_sample_fields(sample)
1094    };
1095    let mut fields = fields.into_iter();
1096    match (fields.next(), fields.next()) {
1097        (Some(x), Some(y)) => (x, y),
1098        _ => placeholder_fields(),
1099    }
1100}
1101
1102/// The Vega-Lite specification the CMS chart block offers, which is the whole answer. It carries no
1103/// data of its own, as a generated specification must not: the teacher's data file is attached to it
1104/// afterwards.
1105fn chart_spec_payload(request: &MockRequest) -> String {
1106    let (x, y) = charted_fields(request);
1107    json!({
1108        "$schema": "https://vega.github.io/schema/vega-lite/v6.json",
1109        "description": "Mock AI generated bar chart",
1110        "mark": "bar",
1111        "encoding": {
1112            "x": {"field": x.name, "type": x.field_type},
1113            "y": {"field": y.name, "type": y.field_type}
1114        }
1115    })
1116    .to_string()
1117}
1118
1119/// The course description summary, in the shape Sisu expects it in.
1120const COURSE_DESCRIPTION_PAYLOAD: &str = r#"{"modules":[{"description":"Introductory course to containers and containerization with Docker. Introduces containerization with Docker and relevant concepts such as image and volume. After completion, students are able to run containerized applications, containerize applications, utilize volumes to store data persistently outside containers, use port mapping to enable access via TCP to containerized applications, and share their own containers publicly. No hard prerequisites; Linux operating systems and web development experience are useful.","prerequisites":["No hard prerequisites","Linux operating systems and web development experience are useful"],"course_code":"TKT21036"}],"audience":["everyone"],"course_description":"Introductory course to containers and containerization with Docker. Introduces containerization with Docker and relevant concepts such as image and volume. After completion, students are able to run containerized applications, containerize applications, utilize volumes to store data persistently outside containers, use port mapping to enable access via TCP to containerized applications, and share their own containers publicly."}"#;
1121
1122const PROMPT_CREATION_PAYLOAD: &str = r#"{"prompt":"You are a helpful, clear, and concise chatbot for a course. Your purpose is to help learners understand and navigate the course, answer questions about its content when information is available, explain chatbot-related concepts at an appropriate level, and support learning with examples or step-by-step guidance. Do not invent course details, lessons, assignments, policies, or resources that have not been provided. If a question cannot be answered from the available information, say so plainly and ask the learner to provide more context or consult the course materials. Be friendly, professional, and focused. Keep responses relevant and avoid overwhelming the learner. When appropriate, suggest a practical next step or ask a clarifying question.","first_message":"Hi! I’m here to help you. Ask me about anything you’d like!","suggested_messages":["Can you pls help me?","Nice weather we're having.","Hello?"]}"#;
1123
1124// GET /api/v0/mock_azure/openai/v1/embeddings
1125// POST /api/v0/mock_azure/openai/v1/embeddings
1126async fn mock_azure_embeddings(
1127    app_conf: web::Data<ApplicationConfiguration>,
1128    payload: web::Json<EmbeddingRequest>,
1129) -> ControllerResult<String> {
1130    assert!(app_conf.test_chatbot && app_conf.test_mode);
1131
1132    if payload.input.iter().any(|s| s.trim().is_empty()) {
1133        return Err(ControllerError::new(
1134            ControllerErrorType::BadRequest,
1135            "input must not be empty".to_string(),
1136            None,
1137        ));
1138    }
1139
1140    let mock_response = EmbeddingResponse {
1141        object: "list".to_string(),
1142        model: "mock-embedder-3-small".to_string(),
1143        usage: EmbeddingResponseUsage {
1144            prompt_tokens: payload.input.len() as i32,
1145            total_tokens: payload.input.len() as i32,
1146        },
1147        data: payload
1148            .input
1149            .iter()
1150            .enumerate()
1151            .map(|(index, _)| Embedding {
1152                index: index as i32,
1153                embedding: vec![0.0; 1536],
1154                object: "embedding".to_string(),
1155            })
1156            .collect(),
1157    };
1158    let res = serde_json::to_string(&mock_response)?;
1159    let token = skip_authorize();
1160    token.authorized_ok(res)
1161}
1162
1163pub fn _add_routes(cfg: &mut ServiceConfig) {
1164    cfg.route(
1165        "/api/projects/test/openai/v1/responses",
1166        web::get().to(mock_azure_chat_responses),
1167    )
1168    .route(
1169        "/api/projects/test/openai/v1/responses",
1170        web::post().to(mock_azure_chat_responses),
1171    )
1172    .route("openai/v1/embeddings", web::get().to(mock_azure_embeddings))
1173    .route(
1174        "openai/v1/embeddings",
1175        web::post().to(mock_azure_embeddings),
1176    );
1177}
1178
1179#[cfg(test)]
1180mod tests {
1181    use headless_lms_chatbot::{
1182        azure_chatbot::azure::protocol::{
1183            AISearchOutput, OutputItem, ReceivedOutputItem, ResponseOutput,
1184        },
1185        chatbot_tools::ClientChatbotTool,
1186        llm_utils::{LLMResponse, parse_text_completion},
1187    };
1188    use regex::Regex;
1189
1190    use super::*;
1191
1192    const BASE_URL: &str = "http://project-331.local";
1193
1194    /// Pairs every `event:` line of a Server-Sent Events body with the `data:` line after it.
1195    fn sse_events(body: &str) -> Vec<(&str, &str)> {
1196        let mut events = Vec::new();
1197        let mut pending = None;
1198        for line in body.lines() {
1199            if let Some(event) = line.strip_prefix("event: ") {
1200                pending = Some(event);
1201            } else if let Some(data) = line.strip_prefix("data: ")
1202                && let Some(event) = pending.take()
1203            {
1204                events.push((event, data));
1205            }
1206        }
1207        events
1208    }
1209
1210    /// The body the mock answers `request` with, through the dispatch the handler runs.
1211    fn respond(request: &MockRequest) -> String {
1212        let scenario = pick_scenario(request).expect("the mock answers this request");
1213        (scenario.respond)(request, BASE_URL)
1214    }
1215
1216    /// Every registered scenario's example body of the given kind, named for failure messages.
1217    fn example_bodies(stream: bool) -> Vec<(&'static str, String)> {
1218        SCENARIOS
1219            .iter()
1220            .filter(|scenario| (scenario.example)().stream == stream)
1221            .map(|scenario| {
1222                let example = (scenario.example)();
1223                (scenario.name, (scenario.respond)(&example, BASE_URL))
1224            })
1225            .collect()
1226    }
1227
1228    /// The tools called by the function call items among `events`.
1229    fn called_tool_names(events: &[(&str, &str)]) -> Vec<String> {
1230        events
1231            .iter()
1232            .filter_map(|(_, data)| {
1233                match serde_json::from_str::<ResponseOutput>(data)
1234                    .ok()?
1235                    .item?
1236                    .known()?
1237                {
1238                    OutputItem::FunctionCall { tool_name, .. } => Some(tool_name),
1239                    _ => None,
1240                }
1241            })
1242            .collect()
1243    }
1244
1245    /// Where the round's first delta is, having checked the lifecycle events every round needs
1246    /// around it: the response id before the first delta, and the terminal event last.
1247    fn first_delta(events: &[(&str, &str)]) -> usize {
1248        let index = events
1249            .iter()
1250            .position(|(event, _)| event.ends_with(".delta"))
1251            .expect("The round streams a delta event");
1252        assert!(
1253            events[..index]
1254                .iter()
1255                .any(|(event, _)| *event == "response.created"),
1256            "The response id has to be known before the first delta"
1257        );
1258        assert_eq!(
1259            events.last().map(|(event, _)| *event),
1260            Some("response.completed")
1261        );
1262        index
1263    }
1264
1265    /// The names of the tools the chatbot runs itself.
1266    fn registered_tool_names() -> Vec<String> {
1267        get_chatbot_tool_definitions()
1268            .into_iter()
1269            .filter_map(|definition| match definition {
1270                AzureLLMToolDefinition::Function(function) => Some(function.name),
1271                AzureLLMToolDefinition::Search(_) => None,
1272            })
1273            .collect()
1274    }
1275
1276    /// The chatbot parses every streamed `data:` line into its own types and kills the whole
1277    /// conversation on one it cannot read, so a mistyped field here is otherwise only visible by
1278    /// running the whole stack.
1279    #[test]
1280    fn every_streamed_data_line_parses_into_chatbot_types() {
1281        let bodies = example_bodies(true);
1282        assert!(!bodies.is_empty(), "no streamed scenario is registered");
1283        for (name, body) in bodies {
1284            let events = sse_events(&body);
1285            assert!(!events.is_empty(), "{name} streams no events");
1286            for (event, data) in events {
1287                let parsed: ResponseOutput = serde_json::from_str(data).unwrap_or_else(|e| {
1288                    panic!("{name} streams a {event} the chatbot cannot parse: {e}\n{data}")
1289                });
1290                if event.starts_with("response.output_item.") {
1291                    assert!(
1292                        parsed.item.is_some(),
1293                        "{name}: {event} carries no item\n{data}"
1294                    );
1295                }
1296                if event == "response.created" {
1297                    assert!(
1298                        parsed.response.and_then(|response| response.id).is_some(),
1299                        "{name}: {event} carries no response id\n{data}"
1300                    );
1301                }
1302            }
1303        }
1304    }
1305
1306    /// A scenario answering an example its own `matches` rejects would leave the round it stands
1307    /// for untested while every test built from the registry passed.
1308    #[test]
1309    fn every_scenario_answers_its_own_example() {
1310        for scenario in SCENARIOS {
1311            let request = (scenario.example)();
1312            let picked = pick_scenario(&request).map(|picked| picked.name);
1313            assert_eq!(
1314                picked,
1315                Some(scenario.name),
1316                "{} is not the scenario its own example is answered by",
1317                scenario.name
1318            );
1319        }
1320    }
1321
1322    /// The structured output features parse the text content again as their own response shape, so
1323    /// both layers have to hold.
1324    #[test]
1325    fn structured_output_responses_parse_into_chatbot_types() {
1326        let bodies = example_bodies(false);
1327        assert!(!bodies.is_empty(), "no blocking scenario is registered");
1328        for (name, body) in bodies {
1329            let completion: LLMResponse = serde_json::from_str(&body)
1330                .unwrap_or_else(|e| panic!("{name} does not parse as an LLM response: {e}"));
1331            let content = parse_text_completion(completion)
1332                .unwrap_or_else(|e| panic!("{name} has no text content: {e}"));
1333            serde_json::from_str::<Value>(&content).unwrap_or_else(|e| {
1334                panic!("{name} content is not the JSON the feature parses: {e}\n{content}")
1335            });
1336        }
1337    }
1338
1339    /// Each feature parses the text content as its own response shape, so a payload nested inside
1340    /// another object would satisfy the test above and still break all three.
1341    #[test]
1342    fn a_blocking_response_carries_its_payload_as_the_whole_text_content() {
1343        for payload in [
1344            MESSAGE_SUGGESTION_PAYLOAD,
1345            CMS_SUGGESTION_PAYLOAD,
1346            COURSE_DESCRIPTION_PAYLOAD,
1347        ] {
1348            let completion: LLMResponse = serde_json::from_str(&blocking_response(payload))
1349                .expect("the blocking response parses as an LLM response");
1350            let content = parse_text_completion(completion).expect("the response has text content");
1351            assert_eq!(content, payload);
1352        }
1353    }
1354
1355    /// A chart spec request whose prompt carries `sample` as the teacher's data file.
1356    fn chart_request_sampling(sample: &str) -> MockRequest {
1357        MockRequest::json_object(&format!(
1358            "{CHART_SPEC_REQUEST}\n\nRequest:\nA bar chart{DATA_SAMPLE_HEADING}{sample}"
1359        ))
1360    }
1361
1362    /// The fields the answer to `request` encodes, as (x, y) pairs of field name and type.
1363    fn charted_encoding(request: &MockRequest) -> ((String, String), (String, String)) {
1364        let spec: Value = serde_json::from_str(&chart_spec_payload(request))
1365            .expect("the chart specification is the whole answer, so it must be JSON");
1366        let channel = |name: &str| {
1367            let channel = &spec["encoding"][name];
1368            (
1369                channel["field"]
1370                    .as_str()
1371                    .expect("the channel names a field")
1372                    .to_string(),
1373                channel["type"]
1374                    .as_str()
1375                    .expect("the channel gives a type")
1376                    .to_string(),
1377            )
1378        };
1379        (channel("x"), channel("y"))
1380    }
1381
1382    /// The chart generator rejects any specification that carries data of its own, so an answer
1383    /// offering one would fail every generation in development and in the system tests.
1384    #[test]
1385    fn the_mock_chart_specification_carries_no_data() {
1386        let spec: Value = serde_json::from_str(&chart_spec_payload(&MockRequest::json_object(
1387            CHART_SPEC_REQUEST,
1388        )))
1389        .expect("the chart specification is the whole answer, so it must be JSON");
1390
1391        assert!(spec.get("data").is_none(), "{spec}");
1392        assert!(spec.get("datasets").is_none(), "{spec}");
1393    }
1394
1395    #[test]
1396    fn the_mock_chart_falls_back_to_placeholder_fields_without_a_sample() {
1397        let (x, y) = charted_encoding(&MockRequest::json_object(CHART_SPEC_REQUEST));
1398
1399        assert_eq!(x, ("category".to_string(), "nominal".to_string()));
1400        assert_eq!(y, ("value".to_string(), "quantitative".to_string()));
1401    }
1402
1403    #[test]
1404    fn the_mock_chart_encodes_the_fields_of_a_json_sample() {
1405        let request = chart_request_sampling(
1406            "[\n  { \"maara\": 0, \"hinta\": 8 },\n  { \"maara\": 50, \"hinta\": 7 }\n]",
1407        );
1408
1409        let (x, y) = charted_encoding(&request);
1410
1411        assert_eq!(x, ("maara".to_string(), "quantitative".to_string()));
1412        assert_eq!(y, ("hinta".to_string(), "quantitative".to_string()));
1413    }
1414
1415    /// serde_json sorts an object's keys, so reading the sample through it would put `apple` on x.
1416    #[test]
1417    fn the_mock_chart_keeps_the_field_order_of_the_sample() {
1418        let request = chart_request_sampling("[{ \"zebra\": \"A\", \"apple\": 1 }]");
1419
1420        let (x, y) = charted_encoding(&request);
1421
1422        assert_eq!(x.0, "zebra");
1423        assert_eq!(y.0, "apple");
1424    }
1425
1426    #[test]
1427    fn the_mock_chart_encodes_the_columns_of_a_csv_sample() {
1428        let request = chart_request_sampling("category,value\nA,28\nB,55\n");
1429
1430        let (x, y) = charted_encoding(&request);
1431
1432        assert_eq!(x, ("category".to_string(), "nominal".to_string()));
1433        assert_eq!(y, ("value".to_string(), "quantitative".to_string()));
1434    }
1435
1436    /// The CMS cuts the sample off at a fixed length, so it can end anywhere in the file.
1437    #[test]
1438    fn the_mock_chart_reads_a_sample_cut_off_mid_row() {
1439        let request =
1440            chart_request_sampling("[{ \"year\": \"2024\", \"sold\": 12 }, { \"year\": \"20");
1441
1442        let (x, y) = charted_encoding(&request);
1443
1444        // Quoted, so a year is a label rather than a number to plot.
1445        assert_eq!(x, ("year".to_string(), "nominal".to_string()));
1446        assert_eq!(y, ("sold".to_string(), "quantitative".to_string()));
1447    }
1448
1449    #[test]
1450    fn the_mock_chart_ignores_a_sample_it_cannot_find_two_fields_in() {
1451        let (x, y) = charted_encoding(&chart_request_sampling("nothing chartable here"));
1452
1453        assert_eq!(x.0, "category");
1454        assert_eq!(y.0, "value");
1455    }
1456
1457    /// A repair round's latest message is the correction, so the sample has to be found further
1458    /// back in the conversation.
1459    #[test]
1460    fn the_mock_chart_reads_the_sample_on_a_repair_round() {
1461        let mut request = chart_request_sampling("category,value\nA,28\n");
1462        request
1463            .conversation
1464            .push("Fix this specification".to_string());
1465        request.message = Some("Fix this specification".to_string());
1466
1467        let (x, _) = charted_encoding(&request);
1468
1469        assert_eq!(x.0, "category");
1470    }
1471
1472    /// The chatbot classifies the round from the function call item it announces and hands the
1473    /// tool parser what follows. The tool parser needs a completed function call and a
1474    /// `response.completed`, and errors on a text delta.
1475    #[test]
1476    fn the_function_call_round_drives_the_tool_call_parser() {
1477        let body = respond(&MockRequest::chat(TOOL_CALL_TRIGGER));
1478        let events = sse_events(&body);
1479
1480        let first_delta = first_delta(&events);
1481        assert_eq!(
1482            events[first_delta].0,
1483            "response.function_call_arguments.delta"
1484        );
1485        assert!(
1486            !events
1487                .iter()
1488                .any(|(event, _)| *event == "response.output_text.delta"),
1489            "A text delta makes the tool parser error out"
1490        );
1491
1492        let called = called_tool_names(&events[first_delta + 1..]);
1493        assert!(
1494            !called.is_empty(),
1495            "The tool parser only sees function calls delivered after the first delta"
1496        );
1497
1498        let registered = registered_tool_names();
1499        for tool_name in called {
1500            assert!(
1501                registered.contains(&tool_name),
1502                "The mock calls {tool_name}, which no chatbot tool is registered under"
1503            );
1504        }
1505    }
1506
1507    /// The round that suspends a turn: the tool it calls has to be one the client answers, and
1508    /// must not be one the chatbot would run itself instead of suspending.
1509    #[test]
1510    fn the_client_tool_call_round_calls_a_tool_only_the_client_answers() {
1511        let body = respond(&MockRequest::chat(CLIENT_TOOL_CALL_TRIGGER));
1512
1513        let called = called_tool_names(&sse_events(&body));
1514        assert!(!called.is_empty(), "The round calls no tool");
1515        let registered = registered_tool_names();
1516        for tool_name in called {
1517            assert!(
1518                tool_is_answered_by_client(&tool_name),
1519                "The mock calls {tool_name} to suspend a turn, but the client does not answer it"
1520            );
1521            assert!(
1522                !registered.contains(&tool_name),
1523                "{tool_name} is registered as a chatbot tool, so the turn would never suspend"
1524            );
1525        }
1526    }
1527
1528    /// The chatbot validates a client tool's arguments before it suspends, so a round the tool
1529    /// would reject never reaches the client: it becomes a failure reported to the LLM instead.
1530    #[test]
1531    fn the_client_tool_call_round_asks_a_question_the_tool_accepts() {
1532        AskMultipleChoiceQuestionTool::parse_arguments(MOCK_MULTIPLE_CHOICE_ARGUMENTS)
1533            .expect("the mock's question passes the tool's own validation");
1534    }
1535
1536    #[test]
1537    fn the_answer_after_a_tool_ran_drives_the_text_parser() {
1538        let body = respond(&MockRequest::after_tool_run());
1539        let events = sse_events(&body);
1540
1541        assert_eq!(events[first_delta(&events)].0, "response.output_text.delta");
1542
1543        let streamed: String = events
1544            .iter()
1545            .filter(|(event, _)| *event == "response.output_text.delta")
1546            .filter_map(|(_, data)| serde_json::from_str::<ResponseOutput>(data).ok()?.delta)
1547            .collect();
1548        assert!(!streamed.is_empty(), "The round streams no text");
1549    }
1550
1551    /// The system tests wait for this answer with the citation markers stripped out by the
1552    /// frontend, and the committed screenshots show it with them rendered as citation pills, so
1553    /// rewording either form here, or shifting a space around a marker, fails the chatbot specs.
1554    #[test]
1555    fn the_search_round_answers_the_text_the_system_tests_wait_for() {
1556        let answer = SEARCH_ANSWER_DELTAS.concat();
1557        assert_eq!(
1558            answer,
1559            "Hello! How can I assist 【0:2†source】 you 【0:1†source】 today?【0:2†source】"
1560        );
1561
1562        // The frontend's REMOVE_CITATIONS_REGEX, which produces the text the specs wait for.
1563        let stripped = Regex::new(r"\s*?【\d+:\d+†source】")
1564            .expect("the citation regex compiles")
1565            .replace_all(&answer, "");
1566        assert_eq!(stripped, "Hello! How can I assist you today?");
1567    }
1568
1569    /// The urls are the only part of the search output the chatbot reads, and it reads them out of
1570    /// a JSON string nested in a JSON string, so the nesting only fails where it is parsed: when
1571    /// the answer's citations are saved.
1572    #[test]
1573    fn the_search_round_streams_document_urls() {
1574        let body = search_and_text_round(BASE_URL);
1575
1576        let search_outputs: Vec<String> = sse_events(&body)
1577            .into_iter()
1578            .filter_map(|(_, data)| {
1579                match serde_json::from_str::<ResponseOutput>(data)
1580                    .ok()?
1581                    .item?
1582                    .known()?
1583                {
1584                    OutputItem::AzureAiSearchCallOutput { output, .. }
1585                        if output.contains("get_urls") =>
1586                    {
1587                        Some(output)
1588                    }
1589                    _ => None,
1590                }
1591            })
1592            .collect();
1593        assert!(
1594            !search_outputs.is_empty(),
1595            "The search round streams no search output with urls"
1596        );
1597        for output in search_outputs {
1598            let parsed: AISearchOutput = serde_json::from_str(&output)
1599                .unwrap_or_else(|e| panic!("The search output does not parse: {e}\n{output}"));
1600            assert_eq!(parsed.get_urls.len(), 3);
1601            for url in parsed.get_urls {
1602                assert!(url.as_str().starts_with(BASE_URL), "{url}");
1603            }
1604        }
1605    }
1606
1607    /// An answer cut short by the token limit ends on `response.incomplete`, never on
1608    /// `response.completed`.
1609    #[test]
1610    fn the_incomplete_answer_round_ends_on_incomplete_not_completed() {
1611        let body = respond(&MockRequest::chat(INCOMPLETE_ANSWER_TRIGGER));
1612        let events = sse_events(&body);
1613        assert_eq!(
1614            events.last().map(|(event, _)| *event),
1615            Some("response.incomplete")
1616        );
1617        assert!(
1618            !events
1619                .iter()
1620                .any(|(event, _)| *event == "response.completed"),
1621            "an incomplete round must not also carry a response.completed"
1622        );
1623    }
1624
1625    /// A stream closed before Azure sends a terminal event must not carry one, or it stops
1626    /// exercising the shape a proxy's clean EOF produces.
1627    #[test]
1628    fn the_truncated_stream_round_carries_no_terminal_event() {
1629        let body = respond(&MockRequest::chat(TRUNCATED_STREAM_TRIGGER));
1630        let events = sse_events(&body);
1631        assert!(!events.is_empty(), "the round streams no events at all");
1632        assert!(
1633            !events
1634                .iter()
1635                .any(|(event, _)| *event == "response.completed" || *event == "response.incomplete"),
1636            "a truncated stream must carry neither response.completed nor response.incomplete"
1637        );
1638    }
1639
1640    /// The completed search output this scenario carries fails to parse as the chatbot's
1641    /// search-output schema, on a `completed` item rather than the in-progress placeholder every
1642    /// other search scenario here uses.
1643    #[test]
1644    fn the_malformed_search_output_round_carries_a_completed_output_that_fails_to_parse() {
1645        let body = respond(&MockRequest::chat(MALFORMED_SEARCH_OUTPUT_TRIGGER));
1646        let events = sse_events(&body);
1647        let output = events
1648            .iter()
1649            .find_map(|(_, data)| {
1650                match serde_json::from_str::<ResponseOutput>(data)
1651                    .ok()?
1652                    .item?
1653                    .known()?
1654                {
1655                    OutputItem::AzureAiSearchCallOutput { output, .. } => Some(output),
1656                    _ => None,
1657                }
1658            })
1659            .expect("the round carries a search output item");
1660        assert!(
1661            serde_json::from_str::<AISearchOutput>(&output).is_err(),
1662            "the output was expected to fail the chatbot's search-output schema: {output}"
1663        );
1664    }
1665
1666    /// An item type the chatbot has no variant for still deserializes as
1667    /// `ReceivedOutputItem::Unreadable` rather than failing the line it arrives on.
1668    #[test]
1669    fn the_unknown_item_type_round_deserializes_as_unreadable() {
1670        let body = respond(&MockRequest::chat(UNKNOWN_ITEM_TYPE_TRIGGER));
1671        let events = sse_events(&body);
1672        let saw_unreadable = events.iter().any(|(event, data)| {
1673            event.starts_with("response.output_item.")
1674                && matches!(
1675                    serde_json::from_str::<ResponseOutput>(data)
1676                        .ok()
1677                        .and_then(|r| r.item),
1678                    Some(ReceivedOutputItem::Unreadable(_))
1679                )
1680        });
1681        assert!(saw_unreadable, "the round carries no unreadable item");
1682    }
1683}