File size: 13,918 Bytes
f15fb1d
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
from __future__ import annotations

import asyncio

import pytest
from app_server_harness import (
    AppServerHarness,
    ev_assistant_message,
    ev_completed,
    ev_completed_with_usage,
    ev_failed,
    ev_response_created,
    sse,
)
from app_server_helpers import (
    agent_message_texts_from_items,
    assistant_message_with_phase,
)

from openai_codex import AsyncCodex, Codex
from openai_codex.generated.v2_all import MessagePhase


def test_sync_thread_run_uses_mock_responses(
    tmp_path,
) -> None:
    """Drive Thread.run through the pinned app-server and inspect the HTTP request."""
    with AppServerHarness(tmp_path) as harness:
        harness.responses.enqueue_assistant_message("Hello from the mock.", response_id="run-1")

        with Codex(config=harness.app_server_config()) as codex:
            thread = codex.thread_start()
            result = thread.run("hello")

        request = harness.responses.single_request()

    body = request.body_json()
    assert {
        "final_response": result.final_response,
        "agent_messages": agent_message_texts_from_items(result.items),
        "has_usage": result.usage is not None,
        "request_model": body["model"],
        "request_stream": body["stream"],
        "request_user_texts": request.message_input_texts("user")[-1:],
    } == {
        "final_response": "Hello from the mock.",
        "agent_messages": ["Hello from the mock."],
        "has_usage": True,
        "request_model": "mock-model",
        "request_stream": True,
        "request_user_texts": ["hello"],
    }


def test_checkout_supports_new_options_and_history_selection(tmp_path) -> None:
    with AppServerHarness(tmp_path) as harness:
        harness.responses.enqueue_assistant_message("Options supported")
        with Codex(config=harness.app_server_config()) as codex:
            thread = codex.thread_start()
            result = thread.run("hello", turn_service_tier="default", source="automation")
            resumed = codex.thread_resume(thread.id, include_turns=False)
            forked = codex.thread_fork(thread.id, include_turns=True)
        assert result.final_response == "Options supported"
        assert resumed.id == thread.id
        assert forked.id != thread.id
        assert harness.responses.single_request().message_input_texts("user")[-1:] == ["hello"]


def test_run_params_and_usage_cross_app_server_boundary(tmp_path) -> None:
    """Thread.run should pass overrides and collect app-server token usage."""
    with AppServerHarness(tmp_path) as harness:
        harness.responses.enqueue_sse(
            sse(
                [
                    ev_response_created("run-overrides"),
                    ev_assistant_message("msg-run-overrides", "overrides applied"),
                    ev_completed_with_usage(
                        "run-overrides",
                        input_tokens=11,
                        cached_input_tokens=3,
                        output_tokens=7,
                        reasoning_output_tokens=5,
                        total_tokens=18,
                    ),
                ]
            )
        )

        with Codex(config=harness.app_server_config()) as codex:
            thread = codex.thread_start()
            result = thread.run(
                "use overrides",
                model="mock-model-override",
            )
            request = harness.responses.single_request()

    usage_payload = None
    if result.usage is not None:
        dumped_usage = result.usage.model_dump(by_alias=True, mode="json")
        usage_payload = {
            "last": dumped_usage["last"],
            "total": dumped_usage["total"],
        }
    assert {
        "final_response": result.final_response,
        "request_model": request.body_json()["model"],
        "usage": usage_payload,
    } == {
        "final_response": "overrides applied",
        "request_model": "mock-model-override",
        "usage": {
            "last": {
                "cacheWriteInputTokens": 0,
                "cachedInputTokens": 3,
                "inputTokens": 11,
                "outputTokens": 7,
                "reasoningOutputTokens": 5,
                "totalTokens": 18,
            },
            "total": {
                "cacheWriteInputTokens": 0,
                "cachedInputTokens": 3,
                "inputTokens": 11,
                "outputTokens": 7,
                "reasoningOutputTokens": 5,
                "totalTokens": 18,
            },
        },
    }


def test_async_thread_run_uses_mock_responses(
    tmp_path,
) -> None:
    """Async Thread.run should exercise the same app-server boundary."""

    async def scenario() -> None:
        """Run the async client against a real app-server process."""
        with AppServerHarness(tmp_path) as harness:
            harness.responses.enqueue_assistant_message(
                "Hello async.",
                response_id="async-run-1",
            )

            async with AsyncCodex(config=harness.app_server_config()) as codex:
                thread = await codex.thread_start()
                result = await thread.run("async hello")

            request = harness.responses.single_request()

        assert {
            "final_response": result.final_response,
            "agent_messages": agent_message_texts_from_items(result.items),
            "request_user_texts": request.message_input_texts("user")[-1:],
        } == {
            "final_response": "Hello async.",
            "agent_messages": ["Hello async."],
            "request_user_texts": ["async hello"],
        }

    asyncio.run(scenario())


def test_sync_turn_result_uses_last_unknown_phase_message(tmp_path) -> None:
    """TurnResult should use the last unknown-phase agent message as final text."""
    with AppServerHarness(tmp_path) as harness:
        harness.responses.enqueue_sse(
            sse(
                [
                    ev_response_created("items-last"),
                    ev_assistant_message("msg-items-first", "First message"),
                    ev_assistant_message("msg-items-second", "Second message"),
                    ev_completed("items-last"),
                ]
            )
        )

        with Codex(config=harness.app_server_config()) as codex:
            result = codex.thread_start().run("case: last unknown phase wins")

    assert {
        "final_response": result.final_response,
        "agent_messages": agent_message_texts_from_items(result.items),
    } == {
        "final_response": "Second message",
        "agent_messages": ["First message", "Second message"],
    }


def test_sync_turn_result_preserves_empty_last_message(tmp_path) -> None:
    """TurnResult should preserve an empty final agent message instead of skipping it."""
    with AppServerHarness(tmp_path) as harness:
        harness.responses.enqueue_sse(
            sse(
                [
                    ev_response_created("items-empty"),
                    ev_assistant_message("msg-items-nonempty", "First message"),
                    ev_assistant_message("msg-items-empty", ""),
                    ev_completed("items-empty"),
                ]
            )
        )

        with Codex(config=harness.app_server_config()) as codex:
            result = codex.thread_start().run("case: empty last message")

    assert {
        "final_response": result.final_response,
        "agent_messages": agent_message_texts_from_items(result.items),
    } == {
        "final_response": "",
        "agent_messages": ["First message", ""],
    }


def test_sync_turn_result_does_not_promote_commentary_only_to_final(tmp_path) -> None:
    """TurnResult final_response should stay unset when app-server marks only commentary."""
    with AppServerHarness(tmp_path) as harness:
        harness.responses.enqueue_sse(
            sse(
                [
                    ev_response_created("items-commentary"),
                    assistant_message_with_phase(
                        "msg-items-commentary",
                        "Commentary",
                        MessagePhase.commentary,
                    ),
                    ev_completed("items-commentary"),
                ]
            )
        )

        with Codex(config=harness.app_server_config()) as codex:
            result = codex.thread_start().run("case: commentary only")

    assert {
        "final_response": result.final_response,
        "agent_messages": agent_message_texts_from_items(result.items),
    } == {
        "final_response": None,
        "agent_messages": ["Commentary"],
    }


def test_async_turn_result_uses_last_unknown_phase_message(tmp_path) -> None:
    """Async TurnResult should use the last unknown-phase agent message."""

    async def scenario() -> None:
        """Run one async result-mapping case against a pinned app-server."""
        with AppServerHarness(tmp_path) as harness:
            harness.responses.enqueue_sse(
                sse(
                    [
                        ev_response_created("async-items-last"),
                        ev_assistant_message(
                            "msg-async-items-first",
                            "First async message",
                        ),
                        ev_assistant_message(
                            "msg-async-items-second",
                            "Second async message",
                        ),
                        ev_completed("async-items-last"),
                    ]
                )
            )

            async with AsyncCodex(config=harness.app_server_config()) as codex:
                result = await (await codex.thread_start()).run("case: async last unknown phase")

        assert {
            "final_response": result.final_response,
            "agent_messages": agent_message_texts_from_items(result.items),
        } == {
            "final_response": "Second async message",
            "agent_messages": ["First async message", "Second async message"],
        }

    asyncio.run(scenario())


def test_async_turn_result_does_not_promote_commentary_only_to_final(
    tmp_path,
) -> None:
    """Async TurnResult final_response should stay unset for commentary-only output."""

    async def scenario() -> None:
        """Run one async commentary mapping case against a pinned app-server."""
        with AppServerHarness(tmp_path) as harness:
            harness.responses.enqueue_sse(
                sse(
                    [
                        ev_response_created("async-items-commentary"),
                        assistant_message_with_phase(
                            "msg-async-items-commentary",
                            "Async commentary",
                            MessagePhase.commentary,
                        ),
                        ev_completed("async-items-commentary"),
                    ]
                )
            )

            async with AsyncCodex(config=harness.app_server_config()) as codex:
                result = await (await codex.thread_start()).run("case: async commentary only")

        assert {
            "final_response": result.final_response,
            "agent_messages": agent_message_texts_from_items(result.items),
        } == {
            "final_response": None,
            "agent_messages": ["Async commentary"],
        }

    asyncio.run(scenario())


def test_thread_run_raises_when_real_app_server_reports_failed_turn(tmp_path) -> None:
    """Thread.run should surface the failed turn error emitted by app-server."""
    with AppServerHarness(tmp_path) as harness:
        harness.responses.enqueue_sse(
            sse(
                [
                    ev_response_created("failed-run"),
                    ev_failed("failed-run", "boom from mock model"),
                ]
            )
        )

        with Codex(config=harness.app_server_config()) as codex:
            thread = codex.thread_start()
            with pytest.raises(RuntimeError, match="boom from mock model"):
                thread.run("trigger failure")


def test_final_answer_phase_survives_real_app_server_mapping(tmp_path) -> None:
    """TurnResult should use the final-answer item emitted by app-server."""
    with AppServerHarness(tmp_path) as harness:
        harness.responses.enqueue_sse(
            sse(
                [
                    ev_response_created("phase-1"),
                    {
                        **ev_assistant_message("msg-commentary", "Commentary"),
                        "item": {
                            **ev_assistant_message("msg-commentary", "Commentary")["item"],
                            "phase": MessagePhase.commentary.value,
                        },
                    },
                    {
                        **ev_assistant_message("msg-final", "Final answer"),
                        "item": {
                            **ev_assistant_message("msg-final", "Final answer")["item"],
                            "phase": MessagePhase.final_answer.value,
                        },
                    },
                    ev_completed("phase-1"),
                ]
            )
        )

        with Codex(config=harness.app_server_config()) as codex:
            result = codex.thread_start().run("choose final answer")

    assert {
        "final_response": result.final_response,
        "items": [
            {
                "text": item.root.text,
                "phase": None if item.root.phase is None else item.root.phase.value,
            }
            for item in result.items
            if item.root.type == "agentMessage"
        ],
    } == {
        "final_response": "Final answer",
        "items": [
            {"text": "Commentary", "phase": MessagePhase.commentary.value},
            {"text": "Final answer", "phase": MessagePhase.final_answer.value},
        ],
    }