hyf

Context-aware query service for Radroots
git clone https://radroots.dev/git/hyf.git
Log | Files | Refs | README | LICENSE

test_provider_adapter.mojo (97585B)


      1 from std.collections import List
      2 from std.ffi import ErrNo
      3 from std.testing import TestSuite, assert_equal, assert_raises, assert_true
      4 
      5 from json import Value, loads
      6 
      7 from hyf_assist.contract import max_local_query_rewrite_route
      8 from hyf_core.request_context import default_request_context
      9 from hyf_provider.client import (
     10     get_max_local_health,
     11     max_local_chat_completions_url,
     12     post_max_local_chat_completion,
     13 )
     14 from hyf_provider.config import (
     15     MaxLocalProviderConfig,
     16     max_local_provider_config_from_runtime,
     17 )
     18 from hyf_provider.health import max_local_health_failure_from_reason
     19 from hyf_provider.max_local import max_local_query_rewrite_failure_from_reason
     20 from hyf_provider.result import parse_query_analysis_from_chat_completion
     21 from hyf_provider.schema import build_query_rewrite_request_body
     22 from hyf_runtime.config import (
     23     HyfAssistedRuntimeConfig,
     24     HyfExecutionRuntimeConfig,
     25     HyfLoadedRuntimeConfig,
     26     HyfMaxLocalProviderRuntimeConfig,
     27     HyfRuntimeConfig,
     28     HyfServiceRuntimeConfig,
     29     default_loaded_runtime_config,
     30 )
     31 from parent_lifecycle import CleanupGuard, now_ms, open_fd_count_checked
     32 from max_local_process_helper import (
     33     reserve_loopback_port,
     34     spawn_max_local_scripted,
     35     spawn_max_local_stub,
     36 )
     37 from bounded_call_helper import (
     38     BoundedCallReport,
     39     account_raw_bytes,
     40     parse_bounded_report,
     41     run_bounded_call,
     42 )
     43 from strict_fixture import (
     44     ExchangeScript,
     45     exchange_script,
     46     is_peer_close_cause,
     47     json_escape,
     48     write_errno_class,
     49 )
     50 
     51 # H007 BC02: each bounded-call invocation carries its own correlation value, so
     52 # a report produced for one call can never be accepted for another.
     53 comptime BOUNDED_CORRELATION_REFUSAL = 101
     54 comptime BOUNDED_CORRELATION_BODY_STALL = 102
     55 comptime BOUNDED_CORRELATION_DELAYED_HEAD = 103
     56 comptime BOUNDED_CORRELATION_DELAYED_SUCCESS = 104
     57 comptime BOUNDED_CORRELATION_RAW_HEAD_BODY = 105
     58 comptime BOUNDED_CORRELATION_BOUNDED_STALL = 106
     59 comptime BOUNDED_CORRELATION_NEVER_RETURN = 107
     60 comptime BOUNDED_CORRELATION_RETAINED_CLEANUP = 108
     61 comptime BOUNDED_CORRELATION_REUSED_DESCRIPTOR = 109
     62 comptime BOUNDED_CORRELATION_MUTANT_EXIT7 = 111
     63 comptime BOUNDED_CORRELATION_MUTANT_UNTERMINATED = 112
     64 comptime BOUNDED_CORRELATION_MUTANT_DUPLICATE = 113
     65 comptime BOUNDED_CORRELATION_MUTANT_DELAYED = 114
     66 comptime BOUNDED_CORRELATION_MUTANT_HUGE = 115
     67 comptime BOUNDED_CORRELATION_WAIT_ERROR = 116
     68 # H007 RP01: distinct correlations for the repaired raw byte-accounting controls.
     69 comptime BOUNDED_CORRELATION_RAW_COALESCED = 117
     70 comptime BOUNDED_CORRELATION_RAW_EMPTY = 118
     71 comptime BOUNDED_CORRELATION_RAW_TRUNCATED = 119
     72 comptime BOUNDED_CORRELATION_RAW_SPLIT_UTF8 = 120
     73 # H007 RP02: distinct correlations for the real-consumer error/retry controls.
     74 comptime BOUNDED_CORRELATION_EARLY_EOF = 121
     75 comptime BOUNDED_CORRELATION_SIGNALED = 122
     76 comptime BOUNDED_CORRELATION_INVALID_UTF8 = 123
     77 comptime BOUNDED_CORRELATION_LATE_EXIT = 124
     78 comptime BOUNDED_CORRELATION_EINTR_RETRY = 125
     79 comptime BOUNDED_CORRELATION_EINTR_DEADLINE = 126
     80 comptime BOUNDED_CORRELATION_POLL_ERROR = 127
     81 # H007 RP03: distinct correlations for the real/synthetic write-error controls.
     82 comptime BOUNDED_CORRELATION_SYNTHETIC_TIMEOUT = 128
     83 comptime BOUNDED_CORRELATION_SYNTHETIC_DESCRIPTOR = 129
     84 comptime BOUNDED_CORRELATION_REAL_ERRNO = 130
     85 comptime BOUNDED_CORRELATION_DECLARED_VS_SYNTHETIC = 131
     86 # H007 RP01: malformed/error raw-observation controls.
     87 comptime BOUNDED_CORRELATION_RAW_EOF = 132
     88 comptime BOUNDED_CORRELATION_RAW_MALFORMED = 133
     89 # H007 OB01-OB03: period-14 observation-integrity controls.
     90 comptime BOUNDED_CORRELATION_OB_UNKNOWN_LENGTH = 140
     91 comptime BOUNDED_CORRELATION_OB_JUNK_LENGTH = 141
     92 comptime BOUNDED_CORRELATION_OB_DUP_LENGTH = 142
     93 comptime BOUNDED_CORRELATION_OB_OVER_CAP = 143
     94 comptime BOUNDED_CORRELATION_OB_SURPLUS = 144
     95 comptime BOUNDED_CORRELATION_OB_NO_LENGTH = 145
     96 comptime BOUNDED_CORRELATION_OB_SPLIT_TERMINATOR = 146
     97 comptime BOUNDED_CORRELATION_OB_SPLIT_BODY = 147
     98 comptime BOUNDED_CORRELATION_OB_NEAR_CAP = 148
     99 comptime BOUNDED_CORRELATION_OB_MALFORMED_HEAD = 149
    100 comptime BOUNDED_CORRELATION_OB_SYNTHETIC = 150
    101 # H007 D44/IL01: distinct correlations for the inclusive raw-body-cap boundary
    102 # controls (declared and EOF-delimited) through the actual acquisition path.
    103 comptime BOUNDED_CORRELATION_OB_CAP_DECLARED_MINUS = 151
    104 comptime BOUNDED_CORRELATION_OB_CAP_DECLARED_EXACT = 152
    105 comptime BOUNDED_CORRELATION_OB_CAP_DECLARED_PLUS = 153
    106 comptime BOUNDED_CORRELATION_OB_CAP_NO_LENGTH_MINUS = 154
    107 comptime BOUNDED_CORRELATION_OB_CAP_NO_LENGTH_EXACT = 155
    108 comptime BOUNDED_CORRELATION_OB_CAP_NO_LENGTH_PLUS = 156
    109 comptime BOUNDED_CORRELATION_OB_CAP_NO_LENGTH_STALL = 157
    110 
    111 from flare.net import SocketAddr
    112 from flare.tcp import TcpStream
    113 
    114 
    115 def _provider_runtime_config() -> HyfLoadedRuntimeConfig:
    116     var runtime = HyfExecutionRuntimeConfig()
    117     runtime.default_execution_mode = "deterministic"
    118     runtime.allow_assisted = True
    119     var assisted = HyfAssistedRuntimeConfig()
    120     assisted.provider = "max_local"
    121     assisted.max_local = HyfMaxLocalProviderRuntimeConfig(
    122         enabled=True,
    123         base_url="http://127.0.0.1:8000/v1/",
    124         health_url="http://127.0.0.1:8000/health",
    125         model="max-local-query-rewrite",
    126         request_timeout_ms=15000,
    127     )
    128     return HyfLoadedRuntimeConfig(
    129         artifact_present=True,
    130         loaded=True,
    131         compiled_defaults_active=False,
    132         load_state="loaded",
    133         load_error="",
    134         effective=HyfRuntimeConfig(
    135             service=HyfServiceRuntimeConfig(transport="stdio"),
    136             runtime=runtime.copy(),
    137             assisted=assisted.copy(),
    138         ),
    139     )
    140 
    141 
    142 def _provider_config() -> MaxLocalProviderConfig:
    143     return MaxLocalProviderConfig(
    144         base_url="http://127.0.0.1:8000/v1/",
    145         health_url="http://127.0.0.1:8000/health",
    146         model="max-local-query-rewrite",
    147         request_timeout_ms=15000,
    148     )
    149 
    150 
    151 def _provider_config_for_port(port: Int) -> MaxLocalProviderConfig:
    152     return MaxLocalProviderConfig(
    153         base_url="http://127.0.0.1:" + String(port) + "/v1/",
    154         health_url="http://127.0.0.1:" + String(port) + "/health",
    155         model="max-local-query-rewrite",
    156         request_timeout_ms=15000,
    157     )
    158 
    159 
    160 def _invalid_base_url_provider_config() -> MaxLocalProviderConfig:
    161     return MaxLocalProviderConfig(
    162         base_url="ftp://127.0.0.1:8000/v1/",
    163         health_url="http://127.0.0.1:8000/health",
    164         model="max-local-query-rewrite",
    165         request_timeout_ms=15000,
    166     )
    167 
    168 
    169 def _invalid_health_url_provider_config() -> MaxLocalProviderConfig:
    170     return MaxLocalProviderConfig(
    171         base_url="http://127.0.0.1:8000/v1/",
    172         health_url="ftp://127.0.0.1:8000/health",
    173         model="max-local-query-rewrite",
    174         request_timeout_ms=15000,
    175     )
    176 
    177 
    178 def _analysis_json_text() -> String:
    179     return (
    180         '{"original_text":"eggs near me",'
    181         '"normalized_text":"eggs near me",'
    182         '"rewritten_text":"eggs",'
    183         '"query_terms":["eggs"],'
    184         '"normalization_signals":["local_intent_detected"],'
    185         '"ranking_hints":["prefer_local_results"],'
    186         '"extracted_filters":{'
    187         '"local_intent":true,'
    188         '"fulfillment":"unspecified",'
    189         '"time_window":"unspecified"'
    190         "}}"
    191     )
    192 
    193 
    194 def _chat_completion_response_with_content(content: String) raises -> Value:
    195     var response = loads("{}")
    196     var choices = loads("[]")
    197     var choice = loads("{}")
    198     var message = loads("{}")
    199     message.set("content", Value(content))
    200     choice.set("message", message)
    201     choices.append(choice)
    202     response.set("choices", choices)
    203     return response^
    204 
    205 
    206 def _chat_completion_response() raises -> Value:
    207     return _chat_completion_response_with_content(_analysis_json_text())
    208 
    209 
    210 def _assert_query_rewrite_failure(
    211     reason: String, expected_kind: String, expected_reason: String
    212 ) raises:
    213     var failure = max_local_query_rewrite_failure_from_reason(reason)
    214     assert_equal(failure.kind, expected_kind)
    215     assert_equal(failure.reason, expected_reason)
    216 
    217 
    218 def _assert_health_failure(
    219     reason: String, expected_kind: String, expected_reason: String
    220 ) raises:
    221     var failure = max_local_health_failure_from_reason(reason)
    222     assert_equal(failure.kind, expected_kind)
    223     assert_equal(failure.reason, expected_reason)
    224 
    225 
    226 def _assert_chat_completion_parse_failure(
    227     response: Value, expected_error: String
    228 ) raises:
    229     try:
    230         _ = parse_query_analysis_from_chat_completion(response)
    231     except e:
    232         assert_equal(String(e), expected_error)
    233         return
    234     raise Error("expected chat completion parse failure")
    235 
    236 
    237 def test_provider_config_maps_runtime_config() raises:
    238     var config = max_local_provider_config_from_runtime(
    239         _provider_runtime_config()
    240     )
    241 
    242     assert_equal(config.base_url, "http://127.0.0.1:8000/v1/")
    243     assert_equal(config.health_url, "http://127.0.0.1:8000/health")
    244     assert_equal(config.model, "max-local-query-rewrite")
    245     assert_equal(config.request_timeout_ms, 15000)
    246 
    247 
    248 def test_max_local_route_is_derived_from_assisted_contract() raises:
    249     assert_equal(
    250         max_local_query_rewrite_route(),
    251         "provider_runtime.query_rewrite.max_local",
    252     )
    253 
    254 
    255 def test_max_local_provider_failure_mapping_preserves_reason_tokens() raises:
    256     _assert_query_rewrite_failure("timeout", "transport", "timeout")
    257     _assert_query_rewrite_failure(
    258         "connection_failed", "transport", "connection_failed"
    259     )
    260     _assert_query_rewrite_failure("invalid_url", "transport", "invalid_url")
    261     _assert_query_rewrite_failure(
    262         "provider_non_2xx", "http_status", "provider_non_2xx"
    263     )
    264     _assert_query_rewrite_failure(
    265         "provider_error_payload",
    266         "provider_payload",
    267         "provider_error_payload",
    268     )
    269     _assert_query_rewrite_failure(
    270         "provider_invalid_json",
    271         "provider_payload",
    272         "provider_invalid_json",
    273     )
    274     _assert_query_rewrite_failure(
    275         "provider_schema_invalid",
    276         "provider_payload",
    277         "provider_schema_invalid",
    278     )
    279     _assert_query_rewrite_failure(
    280         "provider_empty_choices",
    281         "provider_payload",
    282         "provider_empty_choices",
    283     )
    284     _assert_query_rewrite_failure(
    285         "provider_missing_content",
    286         "provider_payload",
    287         "provider_missing_content",
    288     )
    289     _assert_query_rewrite_failure(
    290         "unknown_transport", "provider", "provider_error"
    291     )
    292     _assert_query_rewrite_failure(
    293         "unknown_provider", "provider", "provider_error"
    294     )
    295 
    296 
    297 def test_max_local_health_failure_mapping_preserves_reason_tokens() raises:
    298     _assert_health_failure("timeout", "transport", "timeout")
    299     _assert_health_failure("invalid_url", "transport", "invalid_url")
    300     _assert_health_failure(
    301         "connection_failed", "transport", "connection_failed"
    302     )
    303     _assert_health_failure("non_2xx", "http_status", "non_2xx")
    304     _assert_health_failure(
    305         "unknown_transport", "transport", "connection_failed"
    306     )
    307 
    308 
    309 def test_provider_config_rejects_unconfigured_runtime() raises:
    310     with assert_raises():
    311         _ = max_local_provider_config_from_runtime(
    312             default_loaded_runtime_config()
    313         )
    314 
    315 
    316 def test_max_local_chat_completions_url_trims_base_url() raises:
    317     assert_equal(
    318         max_local_chat_completions_url(_provider_config()),
    319         "http://127.0.0.1:8000/v1/chat/completions",
    320     )
    321 
    322 
    323 def test_max_local_transport_boundary_rejects_invalid_chat_url() raises:
    324     var outcome = post_max_local_chat_completion(
    325         _invalid_base_url_provider_config(), loads("{}")
    326     )
    327 
    328     assert_true(outcome.failure)
    329     assert_true(not outcome.response)
    330     assert_equal(outcome.failure.value().kind, "transport")
    331     assert_equal(outcome.failure.value().reason, "invalid_url")
    332 
    333 
    334 def test_max_local_transport_boundary_rejects_invalid_health_url() raises:
    335     var outcome = get_max_local_health(_invalid_health_url_provider_config())
    336 
    337     assert_true(outcome.failure)
    338     assert_true(not outcome.response)
    339     assert_equal(outcome.failure.value().kind, "transport")
    340     assert_equal(outcome.failure.value().reason, "invalid_url")
    341 
    342 
    343 def test_max_local_transport_boundary_reports_unknown_chat_transport() raises:
    344     var guard_1 = CleanupGuard()
    345     with spawn_max_local_stub(
    346         0, "query_rewrite_malformed_http", 1, guard_1
    347     ) as provider_stub:
    348         var provider_port = provider_stub.port
    349         var outcome = post_max_local_chat_completion(
    350             _provider_config_for_port(provider_port), loads("{}")
    351         )
    352 
    353         assert_true(outcome.failure)
    354         assert_true(not outcome.response)
    355         assert_equal(outcome.failure.value().kind, "transport")
    356         assert_equal(outcome.failure.value().reason, "unknown_transport")
    357 
    358         provider_stub.wait()
    359 
    360     guard_1.assert_clean()
    361 
    362 
    363 def test_max_local_transport_boundary_reports_unknown_health_transport() raises:
    364     var guard_2 = CleanupGuard()
    365     with spawn_max_local_stub(
    366         0, "health_malformed_http", 1, guard_2
    367     ) as provider_stub:
    368         var provider_port = provider_stub.port
    369         var outcome = get_max_local_health(
    370             _provider_config_for_port(provider_port)
    371         )
    372 
    373         assert_true(outcome.failure)
    374         assert_true(not outcome.response)
    375         assert_equal(outcome.failure.value().kind, "transport")
    376         assert_equal(outcome.failure.value().reason, "unknown_transport")
    377 
    378         provider_stub.wait()
    379 
    380     guard_2.assert_clean()
    381 
    382 
    383 def test_query_rewrite_request_body_sets_schema_contract() raises:
    384     var context = default_request_context()
    385     context.return_provenance = True
    386     var body = build_query_rewrite_request_body(
    387         _provider_config(), "eggs near me", context
    388     )
    389 
    390     assert_equal(body["model"].string_value(), "max-local-query-rewrite")
    391     assert_equal(body["messages"][0]["role"].string_value(), "system")
    392     assert_equal(body["messages"][1]["role"].string_value(), "user")
    393     assert_true(
    394         body["messages"][1]["content"].string_value().find("eggs near me") >= 0
    395     )
    396     assert_equal(body["response_format"]["type"].string_value(), "json_schema")
    397     assert_equal(
    398         body["response_format"]["json_schema"]["name"].string_value(),
    399         "query_rewrite",
    400     )
    401     assert_equal(
    402         body["response_format"]["json_schema"]["strict"].bool_value(), True
    403     )
    404     assert_equal(
    405         body["response_format"]["json_schema"]["schema"]["type"].string_value(),
    406         "object",
    407     )
    408 
    409 
    410 def test_chat_completion_response_parses_query_analysis() raises:
    411     var analysis = parse_query_analysis_from_chat_completion(
    412         _chat_completion_response()
    413     )
    414 
    415     assert_equal(analysis.original_text, "eggs near me")
    416     assert_equal(analysis.normalized_text, "eggs near me")
    417     assert_equal(analysis.rewritten_text, "eggs")
    418     assert_equal(len(analysis.query_terms), 1)
    419     assert_equal(analysis.query_terms[0], "eggs")
    420     assert_equal(analysis.extracted_filters.local_intent, True)
    421 
    422 
    423 def test_chat_completion_response_rejects_invalid_json_content() raises:
    424     _assert_chat_completion_parse_failure(
    425         _chat_completion_response_with_content("not json"),
    426         "provider_invalid_json",
    427     )
    428 
    429 
    430 def test_chat_completion_response_rejects_schema_invalid_content() raises:
    431     _assert_chat_completion_parse_failure(
    432         _chat_completion_response_with_content('{"original_text":"eggs"}'),
    433         "provider_schema_invalid",
    434     )
    435 
    436 
    437 def test_chat_completion_response_rejects_empty_choices() raises:
    438     with assert_raises():
    439         _ = parse_query_analysis_from_chat_completion(loads('{"choices":[]}'))
    440 
    441 
    442 def test_chat_completion_response_rejects_top_level_scalar() raises:
    443     with assert_raises():
    444         _ = parse_query_analysis_from_chat_completion(loads('"not object"'))
    445 
    446 
    447 def test_chat_completion_response_rejects_top_level_array() raises:
    448     with assert_raises():
    449         _ = parse_query_analysis_from_chat_completion(loads("[]"))
    450 
    451 
    452 def test_chat_completion_response_rejects_top_level_null() raises:
    453     with assert_raises():
    454         _ = parse_query_analysis_from_chat_completion(loads("null"))
    455 
    456 
    457 def _bounded_timeout_provider_config(
    458     port: Int, timeout_ms: Int
    459 ) -> MaxLocalProviderConfig:
    460     return MaxLocalProviderConfig(
    461         base_url="http://127.0.0.1:" + String(port) + "/v1/",
    462         health_url="http://127.0.0.1:" + String(port) + "/health",
    463         model="max-local-query-rewrite",
    464         request_timeout_ms=timeout_ms,
    465     )
    466 
    467 
    468 def test_provider_refused_connection_is_bounded_and_specific() raises:
    469     # H007/BC01: a refused connection is a bounded, cause-specific transport
    470     # failure, not a hang. The risky product call runs under the parent-enforced
    471     # finite deadline through the shared bounded-call consumer, and it is
    472     # explicitly characterized as a refusal, not as a real connect-timeout
    473     # scenario. No provider client policy is changed here.
    474     var guard = CleanupGuard()
    475     var dead_port = reserve_loopback_port()
    476     var report = run_bounded_call(
    477         "max_local", dead_port, 300, 5000, guard, BOUNDED_CORRELATION_REFUSAL
    478     )
    479     assert_true(report.completed)
    480     assert_true(not report.stopped)
    481     assert_true(report.problem == "")
    482     assert_true(report.domain_failure())
    483     assert_equal(report.outcome, "fail")
    484     assert_equal(report.cause, "transport")
    485     # Current characterized gap: a fast refused connection is not distinguished
    486     # from an unknown transport error, because the elapsed time is below the
    487     # declared request budget.
    488     assert_equal(report.reason, "unknown_transport")
    489     assert_true(report.cleanup_proved)
    490     assert_true(report.elapsed_ms < 5000)
    491     guard.assert_clean()
    492 
    493 
    494 def test_provider_headers_then_stall_is_bounded_and_specific() raises:
    495     # H007: the fixture delivers the response head and then stalls the body.
    496     # Characterized current gap: the declared request timeout does not bound a
    497     # body-read stall, so the response is returned only after the stall
    498     # completes. The control is non-hanging and does not change client policy.
    499     var scripts = List[ExchangeScript]()
    500     var script = exchange_script(
    501         "headers_then_stall",
    502         "POST",
    503         "/v1/chat/completions",
    504         200,
    505         '{"choices":[]}',
    506     )
    507     script.stall_after_head_ms = 1200
    508     scripts.append(script^)
    509     var guard_2 = CleanupGuard()
    510     var timeout_ms = 300
    511     with spawn_max_local_scripted(0, scripts^, guard_2) as provider_stub:
    512         var report = run_bounded_call(
    513             "max_local",
    514             provider_stub.port,
    515             timeout_ms,
    516             5000,
    517             guard_2,
    518             BOUNDED_CORRELATION_BODY_STALL,
    519         )
    520         assert_true(report.ok())
    521         assert_equal(report.status, 200)
    522         assert_true(report.latency_ms >= 1200)
    523         assert_true(report.elapsed_ms < 5000)
    524         provider_stub.wait()
    525     guard_2.assert_clean()
    526 
    527 
    528 def test_provider_delayed_head_is_bounded_and_specific() raises:
    529     # H007: a fixture that delays the whole response beyond the request budget
    530     # is characterized as the same pre-migration gap: the declared timeout does
    531     # not bound the delayed response, which is still returned after it arrives.
    532     # The control is non-hanging and does not change client policy.
    533     var scripts = List[ExchangeScript]()
    534     var script = exchange_script(
    535         "delayed_head",
    536         "POST",
    537         "/v1/chat/completions",
    538         200,
    539         '{"choices":[]}',
    540     )
    541     script.delay_ms = 1200
    542     scripts.append(script^)
    543     var guard_3 = CleanupGuard()
    544     var timeout_ms = 300
    545     with spawn_max_local_scripted(0, scripts^, guard_3) as provider_stub:
    546         var report = run_bounded_call(
    547             "max_local",
    548             provider_stub.port,
    549             timeout_ms,
    550             5000,
    551             guard_3,
    552             BOUNDED_CORRELATION_DELAYED_HEAD,
    553         )
    554         assert_true(report.ok())
    555         assert_equal(report.status, 200)
    556         assert_true(report.latency_ms >= 1200)
    557         assert_true(report.elapsed_ms < 5000)
    558         provider_stub.wait()
    559     guard_3.assert_clean()
    560 
    561 
    562 def _raw_request_text(path: String) -> String:
    563     return (
    564         "POST "
    565         + path
    566         + " HTTP/1.1\r\nhost: 127.0.0.1\r\ncontent-length: 2\r\n"
    567         "connection: close\r\n\r\n{}"
    568     )
    569 
    570 
    571 def _raw_send_then_close(port: Int, path: String) raises:
    572     """A deliberately owned raw client that closes right after its request.
    573 
    574     Used to produce a real peer close during the scripted delayed/body-stall
    575     write without any host or policy change.
    576     """
    577     var client = TcpStream.connect(SocketAddr.localhost(UInt16(port)))
    578     client.write_all(Span[UInt8, _](_raw_request_text(path).as_bytes()))
    579     client.close()
    580 
    581 
    582 def _raw_send_and_read(port: Int, path: String) raises -> String:
    583     var client = TcpStream.connect(SocketAddr.localhost(UInt16(port)))
    584     client.write_all(Span[UInt8, _](_raw_request_text(path).as_bytes()))
    585     var response = String("")
    586     var buffer = InlineArray[Byte, 1024](fill=0)
    587     while True:
    588         var n = client.read(buffer.unsafe_ptr(), 1024)
    589         if n <= 0:
    590             break
    591         response += String(
    592             unsafe_from_utf8=Span(ptr=buffer.unsafe_ptr(), length=Int(n))
    593         )
    594     client.close()
    595     return response^
    596 
    597 
    598 def _delayed_script(label: String, delay_ms: Int) -> ExchangeScript:
    599     var script = exchange_script(
    600         label, "POST", "/v1/chat/completions", 200, '{"choices":[]}'
    601     )
    602     script.delay_ms = delay_ms
    603     return script^
    604 
    605 
    606 def test_provider_strict_delayed_success_under_bounded_harness() raises:
    607     # TC01/TC02: a delayed response is not permission to swallow errors. With a
    608     # client budget above the delay the strict scripted exchange must succeed,
    609     # and the risky call runs under the exact-owned parent-bounded mechanism.
    610     var scripts = List[ExchangeScript]()
    611     var script = _delayed_script("delayed_success", 300)
    612     scripts.append(script^)
    613     var guard = CleanupGuard()
    614     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
    615         var report = run_bounded_call(
    616             "max_local",
    617             provider_stub.port,
    618             5000,
    619             5000,
    620             guard,
    621             BOUNDED_CORRELATION_DELAYED_SUCCESS,
    622         )
    623         assert_true(report.ok())
    624         assert_true(not report.stopped)
    625         assert_equal(report.status, 200)
    626         assert_equal(report.problem, "")
    627         assert_true(report.latency_ms >= 300)
    628         assert_true(report.cleanup_proved)
    629         provider_stub.wait()
    630         assert_true(provider_stub.ok())
    631         assert_equal(provider_stub.request_count(), 1)
    632         assert_equal(provider_stub.connection_count(), 1)
    633     guard.assert_clean()
    634 
    635 
    636 def test_provider_stall_sends_headers_before_body() raises:
    637     # TC02/BC01: prove the fixture delivered the response head before the
    638     # bounded body stall, so the stall control characterizes a body-read stall
    639     # rather than an unrelated connect/refusal condition. The raw header/body
    640     # observation runs through the same parent-bounded consumer as the product
    641     # call, so a hanging peer cannot hang the owning test.
    642     var scripts = List[ExchangeScript]()
    643     var script = exchange_script(
    644         "head_then_stall", "POST", "/v1/chat/completions", 200, '{"choices":[]}'
    645     )
    646     script.stall_after_head_ms = 800
    647     scripts.append(script^)
    648     var guard = CleanupGuard()
    649     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
    650         var report = run_bounded_call(
    651             "raw_head_body",
    652             provider_stub.port,
    653             5000,
    654             5000,
    655             guard,
    656             BOUNDED_CORRELATION_RAW_HEAD_BODY,
    657             "/v1/chat/completions",
    658             "choices",
    659         )
    660         assert_true(report.ok())
    661         assert_equal(report.status, 200)
    662         assert_true(report.head_ms < 400)
    663         assert_true(report.total_ms >= 700)
    664         assert_equal(report.body_match, "yes")
    665         assert_true(report.body_bytes > 0)
    666         provider_stub.wait()
    667         assert_true(provider_stub.ok())
    668     guard.assert_clean()
    669 
    670 
    671 def test_provider_raw_head_body_accounts_coalesced_body() raises:
    672     # RP01: a body that arrives coalesced with the header terminator must be
    673     # accounted as body bytes, never discarded. The scripted exchange writes
    674     # head and body together, so the unchanged raw observer must report the
    675     # exact eight-byte BODYMARK body with a matching declared length.
    676     var scripts = List[ExchangeScript]()
    677     scripts.append(
    678         exchange_script(
    679             "coalesced_body", "POST", "/v1/chat/completions", 200, "BODYMARK"
    680         )
    681     )
    682     var guard = CleanupGuard()
    683     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
    684         var report = run_bounded_call(
    685             "raw_head_body",
    686             provider_stub.port,
    687             5000,
    688             5000,
    689             guard,
    690             BOUNDED_CORRELATION_RAW_COALESCED,
    691             "/v1/chat/completions",
    692             "BODYMARK",
    693         )
    694         assert_true(report.ok())
    695         assert_equal(report.status, 200)
    696         assert_equal(report.body_bytes, 8)
    697         assert_equal(report.body_match, "yes")
    698         assert_equal(report.declared_bytes, 8)
    699         assert_equal(report.length_match, "yes")
    700         provider_stub.wait()
    701         assert_true(provider_stub.ok())
    702     guard.assert_clean()
    703 
    704 
    705 def test_provider_raw_head_body_accounts_empty_body() raises:
    706     # RP01: a declared zero-length body is an exact observation, not a missing
    707     # one: body_bytes is 0 while the declared length still matches.
    708     var scripts = List[ExchangeScript]()
    709     scripts.append(
    710         exchange_script("empty_body", "POST", "/v1/chat/completions", 200, "")
    711     )
    712     var guard = CleanupGuard()
    713     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
    714         var report = run_bounded_call(
    715             "raw_head_body",
    716             provider_stub.port,
    717             5000,
    718             5000,
    719             guard,
    720             BOUNDED_CORRELATION_RAW_EMPTY,
    721             "/v1/chat/completions",
    722             "",
    723         )
    724         assert_true(report.ok())
    725         assert_equal(report.status, 200)
    726         assert_equal(report.body_bytes, 0)
    727         assert_equal(report.body_match, "unknown")
    728         assert_equal(report.declared_bytes, 0)
    729         assert_equal(report.length_match, "yes")
    730         provider_stub.wait()
    731         assert_true(provider_stub.ok())
    732     guard.assert_clean()
    733 
    734 
    735 def test_provider_raw_head_body_reports_truncated_length_mismatch() raises:
    736     # OB02: an explicit truncation control. The peer declares twenty body bytes
    737     # but sends five and closes; the observer must report an explicit
    738     # incomplete observation with the exact five observed bytes and the twenty
    739     # declared, never an unqualified successful call.
    740     var scripts = List[ExchangeScript]()
    741     var script = exchange_script(
    742         "truncated_body", "POST", "/v1/chat/completions", 200, ""
    743     )
    744     script.raw_response = (
    745         "HTTP/1.1 200 OK\r\ncontent-type: application/json\r\n"
    746         "content-length: 20\r\nconnection: close\r\n\r\nSHORT"
    747     )
    748     scripts.append(script^)
    749     var guard = CleanupGuard()
    750     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
    751         var report = run_bounded_call(
    752             "raw_head_body",
    753             provider_stub.port,
    754             5000,
    755             5000,
    756             guard,
    757             BOUNDED_CORRELATION_RAW_TRUNCATED,
    758             "/v1/chat/completions",
    759             "SHORT",
    760         )
    761         assert_true(report.completed)
    762         assert_true(report.domain_failure())
    763         assert_equal(report.cause, "raw_body_incomplete")
    764         assert_equal(report.status, 0)
    765         assert_equal(report.body_bytes, 5)
    766         assert_equal(report.body_match, "yes")
    767         assert_equal(report.declared_bytes, 20)
    768         assert_equal(report.length_match, "no")
    769         assert_equal(report.surplus_bytes, 0)
    770         provider_stub.wait()
    771         assert_true(provider_stub.ok())
    772     guard.assert_clean()
    773 
    774 
    775 def test_provider_raw_head_body_split_utf8_body_is_preserved() raises:
    776     # RP01: a multi-byte body is preserved whole. The observer accumulates
    777     # bytes and searches them directly, so a split character can never be
    778     # misread as a read/packet boundary.
    779     var scripts = List[ExchangeScript]()
    780     var script = exchange_script(
    781         "split_utf8_body",
    782         "POST",
    783         "/v1/chat/completions",
    784         200,
    785         '{"mark":"a☃b"}',
    786     )
    787     script.stall_after_head_ms = 250
    788     scripts.append(script^)
    789     var guard = CleanupGuard()
    790     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
    791         var report = run_bounded_call(
    792             "raw_head_body",
    793             provider_stub.port,
    794             5000,
    795             5000,
    796             guard,
    797             BOUNDED_CORRELATION_RAW_SPLIT_UTF8,
    798             "/v1/chat/completions",
    799             "a☃b",
    800         )
    801         assert_true(report.ok())
    802         assert_equal(report.status, 200)
    803         assert_equal(report.body_bytes, '{"mark":"a☃b"}'.byte_length())
    804         assert_equal(report.body_match, "yes")
    805         assert_equal(report.length_match, "yes")
    806         provider_stub.wait()
    807         assert_true(provider_stub.ok())
    808     guard.assert_clean()
    809 
    810 
    811 def test_raw_accounting_preserves_split_utf8_bytes() raises:
    812     # RP01: the byte accounting itself is chunk-independent. The same bytes are
    813     # supplied as two pieces whose boundary falls inside the three-byte
    814     # character, and the observation is still exact.
    815     var head = "HTTP/1.1 200 OK\r\ncontent-length: 5\r\n\r\n"
    816     var body = "a☃b"
    817     var raw = List[UInt8]()
    818     for byte in head.as_bytes():
    819         raw.append(UInt8(Int(byte)))
    820     for byte in body.as_bytes():
    821         raw.append(UInt8(Int(byte)))
    822     assert_equal(len(raw), head.byte_length() + body.byte_length())
    823     var boundary = head.byte_length() + 3
    824     var split = List[UInt8]()
    825     for index in range(boundary):
    826         split.append(raw[index])
    827     for index in range(boundary, len(raw)):
    828         split.append(raw[index])
    829     var accounting = account_raw_bytes(split^, "☃")
    830     assert_equal(accounting.problem, "")
    831     assert_equal(accounting.status, 200)
    832     assert_equal(accounting.body_bytes, 5)
    833     assert_equal(accounting.body_match, "yes")
    834     assert_equal(accounting.declared_bytes, 5)
    835     assert_equal(accounting.length_match, "yes")
    836 
    837 
    838 def test_raw_accounting_reports_incomplete_head() raises:
    839     # RP01: a buffer with no header terminator is an explicit incomplete-head
    840     # observation, never a completed call with invented fields.
    841     var raw = List[UInt8]()
    842     for byte in "HTTP/1.1 200 OK\r\ncontent-length: 3".as_bytes():
    843         raw.append(UInt8(Int(byte)))
    844     var accounting = account_raw_bytes(raw^, "abc")
    845     assert_equal(accounting.problem, "raw_head_incomplete")
    846     assert_equal(accounting.body_bytes, 0)
    847     assert_equal(accounting.status, 0)
    848 
    849 
    850 def test_raw_accounting_reports_undecodable_head() raises:
    851     # RP01: an undecodable header block is a bounded decode failure, not a
    852     # silently accepted status/body observation.
    853     var raw = List[UInt8]()
    854     for byte in "HTTP/1.1 200 OK\r\nx: ".as_bytes():
    855         raw.append(UInt8(Int(byte)))
    856     raw.append(UInt8(0xFF))
    857     for byte in "\r\n\r\n".as_bytes():
    858         raw.append(UInt8(Int(byte)))
    859     var accounting = account_raw_bytes(raw^, "")
    860     assert_equal(accounting.problem, "raw_head_decode")
    861     assert_equal(accounting.status, 0)
    862 
    863 
    864 def test_provider_raw_head_body_reports_peer_close_without_head() raises:
    865     # RP01: a peer that closes before sending any response head is a bounded
    866     # raw_eof domain failure, not a hang and not an invented status.
    867     var scripts = List[ExchangeScript]()
    868     var script = exchange_script(
    869         "close_without_head", "POST", "/v1/chat/completions", 200, "{}"
    870     )
    871     script.close_before_response = True
    872     scripts.append(script^)
    873     var guard = CleanupGuard()
    874     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
    875         var report = run_bounded_call(
    876             "raw_head_body",
    877             provider_stub.port,
    878             5000,
    879             5000,
    880             guard,
    881             BOUNDED_CORRELATION_RAW_EOF,
    882             "/v1/chat/completions",
    883             "x",
    884         )
    885         assert_true(report.completed)
    886         assert_true(report.domain_failure())
    887         assert_equal(report.cause, "raw_eof")
    888         provider_stub.wait()
    889         assert_true(provider_stub.ok())
    890     guard.assert_clean()
    891 
    892 
    893 def test_provider_raw_head_body_reports_malformed_status() raises:
    894     # RP01: a malformed response with a terminator but no HTTP status line must
    895     # be reported as a bounded malformed observation (status_missing), never
    896     # read as a successful response with an invented status.
    897     var guard = CleanupGuard()
    898     with spawn_max_local_stub(
    899         0, "query_rewrite_malformed_http", 1, guard
    900     ) as provider_stub:
    901         var report = run_bounded_call(
    902             "raw_head_body",
    903             provider_stub.port,
    904             5000,
    905             5000,
    906             guard,
    907             BOUNDED_CORRELATION_RAW_MALFORMED,
    908             "/v1/chat/completions",
    909             "x",
    910         )
    911         assert_true(report.completed)
    912         assert_true(report.domain_failure())
    913         assert_equal(report.cause, "raw_status_missing")
    914         provider_stub.wait()
    915     guard.assert_clean()
    916 
    917 
    918 def test_provider_scripted_permitted_peer_close_is_declared() raises:
    919     # TC01: a script may declare an expected peer close with an exact bounded
    920     # cause and phase. The fixture verifies that declaration and records the
    921     # observed outcome; it no longer tolerates arbitrary write errors.
    922     var scripts = List[ExchangeScript]()
    923     var script = exchange_script(
    924         "permitted_close", "POST", "/v1/chat/completions", 200, '{"choices":[]}'
    925     )
    926     script.stall_after_head_ms = 300
    927     script.expect_peer_close = True
    928     script.expected_close_cause = "broken_pipe"
    929     script.expected_close_phase = "body_stall"
    930     scripts.append(script^)
    931     var guard = CleanupGuard()
    932     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
    933         _raw_send_then_close(provider_stub.port, "/v1/chat/completions")
    934         provider_stub.wait()
    935         assert_true(provider_stub.ok())
    936         assert_true(provider_stub.failure_case().find("_peer_close_") >= 0)
    937         assert_true(provider_stub.failure_case().find("_body_stall") >= 0)
    938     guard.assert_clean()
    939 
    940 
    941 def test_provider_scripted_unexpected_peer_close_fails() raises:
    942     # TC01: an ordinary delayed/stalled script that never declared an expected
    943     # close must fail with a bounded, cause-specific reason instead of
    944     # swallowing the write error.
    945     var scripts = List[ExchangeScript]()
    946     var script = exchange_script(
    947         "unexpected_close",
    948         "POST",
    949         "/v1/chat/completions",
    950         200,
    951         '{"choices":[]}',
    952     )
    953     script.stall_after_head_ms = 300
    954     scripts.append(script^)
    955     var guard = CleanupGuard()
    956     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
    957         _raw_send_then_close(provider_stub.port, "/v1/chat/completions")
    958         provider_stub.reap()
    959         assert_true(not provider_stub.ok())
    960         assert_equal(provider_stub.phase(), "peer_close")
    961         assert_true(provider_stub.reason().find("unexpected_write_") >= 0)
    962         assert_true(provider_stub.reason().find("_body_stall") >= 0)
    963     guard.assert_clean()
    964 
    965 
    966 def test_provider_scripted_wrong_phase_close_fails() raises:
    967     # TC01: a declaration whose phase does not match the observed phase is
    968     # rejected, so a close in the wrong phase can never pass as expected.
    969     var scripts = List[ExchangeScript]()
    970     var script = exchange_script(
    971         "wrong_phase", "POST", "/v1/chat/completions", 200, '{"choices":[]}'
    972     )
    973     script.stall_after_head_ms = 300
    974     script.expect_peer_close = True
    975     script.expected_close_cause = "broken_pipe"
    976     script.expected_close_phase = "delayed_write"
    977     scripts.append(script^)
    978     var guard = CleanupGuard()
    979     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
    980         _raw_send_then_close(provider_stub.port, "/v1/chat/completions")
    981         provider_stub.reap()
    982         assert_true(not provider_stub.ok())
    983         assert_true(provider_stub.reason().find("_body_stall") >= 0)
    984     guard.assert_clean()
    985 
    986 
    987 def test_provider_scripted_injected_write_error_fails() raises:
    988     # TC01: an injected write error is never a peer close, so it must not be
    989     # accepted even when a peer close was declared.
    990     var scripts = List[ExchangeScript]()
    991     var script = exchange_script(
    992         "injected_error", "POST", "/v1/chat/completions", 200, '{"choices":[]}'
    993     )
    994     script.stall_after_head_ms = 200
    995     script.expect_peer_close = True
    996     script.expected_close_cause = "broken_pipe"
    997     script.expected_close_phase = "body_stall"
    998     script.inject_write_error = "Timeout"
    999     scripts.append(script^)
   1000     var guard = CleanupGuard()
   1001     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   1002         _ = _raw_send_and_read(provider_stub.port, "/v1/chat/completions")
   1003         provider_stub.reap()
   1004         assert_true(not provider_stub.ok())
   1005         assert_equal(provider_stub.phase(), "peer_close")
   1006         assert_true(
   1007             provider_stub.reason().find(
   1008                 "unexpected_write_unrelated_error_body_stall"
   1009             )
   1010             >= 0
   1011         )
   1012     guard.assert_clean()
   1013 
   1014 
   1015 def test_provider_scripted_injected_invalid_descriptor_fails() raises:
   1016     # TC01: an injected invalid-descriptor write error must also fail rather
   1017     # than being accepted as a declared peer close.
   1018     var scripts = List[ExchangeScript]()
   1019     var script = exchange_script(
   1020         "injected_descriptor",
   1021         "POST",
   1022         "/v1/chat/completions",
   1023         200,
   1024         '{"choices":[]}',
   1025     )
   1026     script.stall_after_head_ms = 200
   1027     script.expect_peer_close = True
   1028     script.expected_close_cause = "broken_pipe"
   1029     script.expected_close_phase = "body_stall"
   1030     script.inject_write_error = "Bad file descriptor"
   1031     scripts.append(script^)
   1032     var guard = CleanupGuard()
   1033     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   1034         _ = _raw_send_and_read(provider_stub.port, "/v1/chat/completions")
   1035         provider_stub.reap()
   1036         assert_true(not provider_stub.ok())
   1037         assert_equal(provider_stub.phase(), "peer_close")
   1038         assert_true(
   1039             provider_stub.reason().find(
   1040                 "unexpected_write_unrelated_error_body_stall"
   1041             )
   1042             >= 0
   1043         )
   1044     guard.assert_clean()
   1045 
   1046 
   1047 def _realerrno_from_reason(reason: String) raises -> Int:
   1048     """Raw errno recorded in a bounded write-failure reason, or -1."""
   1049     var marker = "realerrno"
   1050     var at = reason.find(marker)
   1051     if at < 0:
   1052         return -1
   1053     var digits = String(reason[byte = at + marker.byte_length() :])
   1054     var end = digits.find("_")
   1055     if end >= 0:
   1056         digits = String(digits[byte=0:end])
   1057     if digits.byte_length() == 0:
   1058         return -1
   1059     return Int(digits)
   1060 
   1061 
   1062 def test_provider_scripted_synthetic_write_timeout_is_not_peer_close() raises:
   1063     # RP03: a narrowly scoped syscall-result seam mapped exactly to the flare
   1064     # write API's EAGAIN/EWOULDBLOCK rendering. The synthetic failure exercises
   1065     # the actual classification and serve rejection path, is labelled synthdecl
   1066     # and can never be accepted as a peer close.
   1067     var scripts = List[ExchangeScript]()
   1068     var script = exchange_script(
   1069         "synthetic_timeout",
   1070         "POST",
   1071         "/v1/chat/completions",
   1072         200,
   1073         '{"choices":[]}',
   1074     )
   1075     script.stall_after_head_ms = 200
   1076     script.inject_write_errno = Int(ErrNo.EAGAIN.value)
   1077     scripts.append(script^)
   1078     var guard = CleanupGuard()
   1079     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   1080         _ = _raw_send_and_read(provider_stub.port, "/v1/chat/completions")
   1081         provider_stub.reap()
   1082         assert_true(not provider_stub.ok())
   1083         assert_equal(provider_stub.phase(), "peer_close")
   1084         assert_true(not is_peer_close_cause("write_timeout"))
   1085         assert_true(
   1086             provider_stub.reason().find(
   1087                 "unexpected_write_write_timeout_body_stall_synthdecl"
   1088             )
   1089             >= 0
   1090         )
   1091     guard.assert_clean()
   1092 
   1093 
   1094 def test_provider_scripted_synthetic_invalid_descriptor_is_not_peer_close() raises:
   1095     # RP03: the same seam mapped to the flare write API's EBADF rendering. The
   1096     # invalid descriptor is classified as invalid_descriptor, labelled synthetic
   1097     # and rejected even when a peer close was declared.
   1098     var scripts = List[ExchangeScript]()
   1099     var script = exchange_script(
   1100         "synthetic_descriptor",
   1101         "POST",
   1102         "/v1/chat/completions",
   1103         200,
   1104         '{"choices":[]}',
   1105     )
   1106     script.stall_after_head_ms = 200
   1107     script.inject_write_errno = Int(ErrNo.EBADF.value)
   1108     script.expect_peer_close = True
   1109     script.expected_close_cause = "broken_pipe"
   1110     script.expected_close_phase = "body_stall"
   1111     scripts.append(script^)
   1112     var guard = CleanupGuard()
   1113     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   1114         _ = _raw_send_and_read(provider_stub.port, "/v1/chat/completions")
   1115         provider_stub.reap()
   1116         assert_true(not provider_stub.ok())
   1117         assert_equal(provider_stub.phase(), "peer_close")
   1118         assert_true(
   1119             provider_stub.reason().find(
   1120                 "unexpected_write_invalid_descriptor_body_stall_synthdecl"
   1121             )
   1122             >= 0
   1123         )
   1124     guard.assert_clean()
   1125 
   1126 
   1127 def test_provider_scripted_declared_peer_close_rejects_synthetic_timeout() raises:
   1128     # RP03: declaring a real peer close does not waive a synthetic timeout. The
   1129     # declared cause is compared against the observed class, so a non-peer-close
   1130     # failure stays a failure.
   1131     var scripts = List[ExchangeScript]()
   1132     var script = exchange_script(
   1133         "declared_vs_synthetic",
   1134         "POST",
   1135         "/v1/chat/completions",
   1136         200,
   1137         '{"choices":[]}',
   1138     )
   1139     script.stall_after_head_ms = 200
   1140     script.expect_peer_close = True
   1141     script.expected_close_cause = "broken_pipe"
   1142     script.expected_close_phase = "body_stall"
   1143     script.inject_write_errno = Int(ErrNo.EAGAIN.value)
   1144     scripts.append(script^)
   1145     var guard = CleanupGuard()
   1146     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   1147         _ = _raw_send_and_read(provider_stub.port, "/v1/chat/completions")
   1148         provider_stub.reap()
   1149         assert_true(not provider_stub.ok())
   1150         assert_true(provider_stub.reason().find("write_timeout") >= 0)
   1151     guard.assert_clean()
   1152 
   1153 
   1154 def test_provider_scripted_real_peer_close_records_raw_errno() raises:
   1155     # RP03: the real (non-synthetic) write failure. A real peer close makes the
   1156     # fixture's real send(2) fail; the fixture records the raw errno alongside
   1157     # the classification, and the errno's class must equal the observed class.
   1158     var scripts = List[ExchangeScript]()
   1159     var script = exchange_script(
   1160         "real_errno", "POST", "/v1/chat/completions", 200, '{"choices":[]}'
   1161     )
   1162     script.stall_after_head_ms = 300
   1163     script.expect_peer_close = True
   1164     script.expected_close_cause = "broken_pipe"
   1165     script.expected_close_phase = "body_stall"
   1166     scripts.append(script^)
   1167     var guard = CleanupGuard()
   1168     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   1169         _raw_send_then_close(provider_stub.port, "/v1/chat/completions")
   1170         provider_stub.wait()
   1171         assert_true(provider_stub.ok())
   1172         var token = provider_stub.failure_case()
   1173         assert_true(token.find("realerrno") >= 0)
   1174         var errno = _realerrno_from_reason(token)
   1175         assert_true(errno > 0)
   1176         var observed = write_errno_class(errno)
   1177         assert_true(observed == "broken_pipe" or observed == "peer_reset")
   1178         assert_true(token.find("_peer_close_" + observed + "_") >= 0)
   1179     guard.assert_clean()
   1180 
   1181 
   1182 def test_provider_scripted_declared_non_peer_cause_is_rejected() raises:
   1183     # TC01: the declared expected cause is restricted to a real peer-close class,
   1184     # so a write timeout can never be waived by declaring it as expected.
   1185     var scripts = List[ExchangeScript]()
   1186     var script = exchange_script(
   1187         "declared_timeout",
   1188         "POST",
   1189         "/v1/chat/completions",
   1190         200,
   1191         '{"choices":[]}',
   1192     )
   1193     script.stall_after_head_ms = 200
   1194     script.expect_peer_close = True
   1195     script.expected_close_cause = "write_timeout"
   1196     script.expected_close_phase = "body_stall"
   1197     script.inject_write_error = "Timeout"
   1198     scripts.append(script^)
   1199     var guard = CleanupGuard()
   1200     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   1201         _ = _raw_send_and_read(provider_stub.port, "/v1/chat/completions")
   1202         provider_stub.reap()
   1203         assert_true(not provider_stub.ok())
   1204         assert_equal(provider_stub.phase(), "declaration")
   1205         assert_true(
   1206             provider_stub.reason().find("invalid_expected_close_declaration")
   1207             >= 0
   1208         )
   1209     guard.assert_clean()
   1210 
   1211 
   1212 def test_provider_scripted_invalid_phase_declaration_is_rejected() raises:
   1213     # EC01: an unknown/empty expected-close phase is an invalid declaration and
   1214     # cannot be satisfied by any real write step. It is rejected before any
   1215     # response work even though the response write would have succeeded.
   1216     var scripts = List[ExchangeScript]()
   1217     var script = exchange_script(
   1218         "invalid_phase",
   1219         "POST",
   1220         "/v1/chat/completions",
   1221         200,
   1222         '{"choices":[]}',
   1223     )
   1224     script.expect_peer_close = True
   1225     script.expected_close_cause = "broken_pipe"
   1226     script.expected_close_phase = "unknown_phase"
   1227     scripts.append(script^)
   1228     var guard = CleanupGuard()
   1229     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   1230         _ = _raw_send_and_read(provider_stub.port, "/v1/chat/completions")
   1231         provider_stub.reap()
   1232         assert_true(not provider_stub.ok())
   1233         assert_equal(provider_stub.phase(), "declaration")
   1234         assert_true(
   1235             provider_stub.reason().find("invalid_expected_close_declaration")
   1236             >= 0
   1237         )
   1238     guard.assert_clean()
   1239 
   1240 
   1241 def test_provider_scripted_missing_expected_close_fails() raises:
   1242     # EC01: the script declares an expected peer close but the response write
   1243     # succeeds, so the declared event never happened. A successful write is not
   1244     # permission to accept a declared close that was not observed.
   1245     var scripts = List[ExchangeScript]()
   1246     var script = exchange_script(
   1247         "missing_expected_close",
   1248         "POST",
   1249         "/v1/chat/completions",
   1250         200,
   1251         '{"choices":[]}',
   1252     )
   1253     script.expect_peer_close = True
   1254     script.expected_close_cause = "broken_pipe"
   1255     script.expected_close_phase = "delayed_write"
   1256     scripts.append(script^)
   1257     var guard = CleanupGuard()
   1258     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   1259         _ = _raw_send_and_read(provider_stub.port, "/v1/chat/completions")
   1260         provider_stub.reap()
   1261         assert_true(not provider_stub.ok())
   1262         assert_equal(provider_stub.phase(), "peer_close")
   1263         assert_true(
   1264             provider_stub.reason().find("missing_expected_close_delayed_write")
   1265             >= 0
   1266         )
   1267     guard.assert_clean()
   1268 
   1269 
   1270 def test_provider_permitted_close_continues_script_sequence() raises:
   1271     # EC01: a permitted peer close consumes that exchange only. The remaining
   1272     # scripted exchange must still be served and counted, so a permitted close
   1273     # can never terminate the whole sequence as successful with unused
   1274     # exchanges.
   1275     var scripts = List[ExchangeScript]()
   1276     var closing = exchange_script(
   1277         "permitted_then_next",
   1278         "POST",
   1279         "/v1/chat/completions",
   1280         200,
   1281         '{"choices":[]}',
   1282     )
   1283     closing.stall_after_head_ms = 300
   1284     closing.expect_peer_close = True
   1285     closing.expected_close_cause = "broken_pipe"
   1286     closing.expected_close_phase = "body_stall"
   1287     scripts.append(closing^)
   1288     scripts.append(
   1289         exchange_script(
   1290             "after_permitted_close",
   1291             "POST",
   1292             "/v1/chat/completions",
   1293             200,
   1294             '{"choices":[]}',
   1295         )
   1296     )
   1297     var guard = CleanupGuard()
   1298     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   1299         _raw_send_then_close(provider_stub.port, "/v1/chat/completions")
   1300         var second = _raw_send_and_read(
   1301             provider_stub.port, "/v1/chat/completions"
   1302         )
   1303         provider_stub.wait()
   1304         assert_true(provider_stub.ok())
   1305         assert_equal(provider_stub.request_count(), 2)
   1306         assert_equal(provider_stub.connection_count(), 2)
   1307         assert_true(provider_stub.failure_case().find("_peer_close_") >= 0)
   1308         assert_true(second.find("choices") >= 0)
   1309     guard.assert_clean()
   1310 
   1311 
   1312 def test_provider_stalled_call_is_parent_bounded() raises:
   1313     # TC02: a real risky client call against a stalling peer runs under the
   1314     # parent deadline and is stopped/reaped by the parent instead of hanging.
   1315     var scripts = List[ExchangeScript]()
   1316     var script = exchange_script(
   1317         "bounded_stall", "POST", "/v1/chat/completions", 200, '{"choices":[]}'
   1318     )
   1319     script.stall_after_head_ms = 3000
   1320     scripts.append(script^)
   1321     var guard = CleanupGuard()
   1322     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   1323         var report = run_bounded_call(
   1324             "max_local",
   1325             provider_stub.port,
   1326             5000,
   1327             600,
   1328             guard,
   1329             BOUNDED_CORRELATION_BOUNDED_STALL,
   1330         )
   1331         assert_true(report.stopped)
   1332         assert_true(not report.completed)
   1333         assert_true(report.cleanup_proved)
   1334         assert_true(report.elapsed_ms >= 500)
   1335         assert_true(report.elapsed_ms < 3000)
   1336         provider_stub.wait()
   1337     guard.assert_clean()
   1338 
   1339 
   1340 def test_provider_never_returning_call_is_stopped_and_reaped() raises:
   1341     # TC02: a deliberate never-returning control proves the parent can stop and
   1342     # reap the call; this is the bounded-harness proof an elapsed assertion
   1343     # after a synchronous call cannot provide.
   1344     var guard = CleanupGuard()
   1345     var report = run_bounded_call(
   1346         "never_return", 0, 300, 400, guard, BOUNDED_CORRELATION_NEVER_RETURN
   1347     )
   1348     assert_true(report.stopped)
   1349     assert_true(not report.completed)
   1350     assert_true(report.cleanup_proved)
   1351     assert_true(report.elapsed_ms >= 350)
   1352     assert_true(report.elapsed_ms < 5000)
   1353     guard.assert_clean()
   1354 
   1355 
   1356 def test_bounded_call_retained_cleanup_recovers_and_leaks_nothing() raises:
   1357     # BC03: an unproved bounded-call cleanup must retain usable ownership in the
   1358     # caller-held guard, recover the exact owned child on retry, and leave no
   1359     # descriptor or child behind — including a following call whose pipe reuses
   1360     # the descriptor number the recovery released.
   1361     var guard = CleanupGuard()
   1362     var fd_before = open_fd_count_checked()
   1363     var report = run_bounded_call(
   1364         "never_return",
   1365         0,
   1366         300,
   1367         400,
   1368         guard,
   1369         BOUNDED_CORRELATION_RETAINED_CLEANUP,
   1370         "/v1/chat/completions",
   1371         "",
   1372         1,
   1373         0,
   1374     )
   1375     assert_true(report.stopped)
   1376     assert_true(not report.completed)
   1377     assert_true(not report.cleanup_proved)
   1378     assert_true(guard.retained() >= 1)
   1379     assert_equal(guard.recover_all(), 0)
   1380     guard.assert_clean()
   1381     var follow = run_bounded_call(
   1382         "never_return",
   1383         0,
   1384         300,
   1385         400,
   1386         guard,
   1387         BOUNDED_CORRELATION_REUSED_DESCRIPTOR,
   1388     )
   1389     assert_true(follow.stopped)
   1390     assert_true(follow.cleanup_proved)
   1391     guard.assert_clean()
   1392     assert_equal(open_fd_count_checked(), fd_before)
   1393     assert_equal(guard.pending(), 0)
   1394 
   1395 
   1396 def test_bounded_call_rejects_nonzero_exit_after_valid_report() raises:
   1397     # BC02 before/after control: the period-12 child-only mutation produced a
   1398     # valid-looking report followed by exit 7 and the old consumer called it
   1399     # completed. The unchanged parent consumer must reject the non-zero exit
   1400     # after observing the natural exit, never kill the child and call it done.
   1401     var guard = CleanupGuard()
   1402     var report = run_bounded_call(
   1403         "mutate_exit7",
   1404         0,
   1405         300,
   1406         500,
   1407         guard,
   1408         BOUNDED_CORRELATION_MUTANT_EXIT7,
   1409     )
   1410     assert_true(not report.completed)
   1411     assert_true(not report.stopped)
   1412     assert_equal(report.problem, "child_exit_7")
   1413     assert_true(report.child_status.find("exited=7") >= 0)
   1414     assert_true(report.cleanup_proved)
   1415     guard.assert_clean()
   1416 
   1417 
   1418 def test_bounded_call_rejects_unterminated_report() raises:
   1419     # BC02 before/after control: an unterminated report is a harness failure,
   1420     # never a completed call.
   1421     var guard = CleanupGuard()
   1422     var report = run_bounded_call(
   1423         "mutate_unterminated",
   1424         0,
   1425         300,
   1426         500,
   1427         guard,
   1428         BOUNDED_CORRELATION_MUTANT_UNTERMINATED,
   1429     )
   1430     assert_true(not report.completed)
   1431     assert_equal(report.problem, "report_unterminated")
   1432     assert_true(report.cleanup_proved)
   1433     guard.assert_clean()
   1434 
   1435 
   1436 def test_bounded_call_rejects_duplicate_report() raises:
   1437     # BC02 before/after control: a second report line is surplus through EOF and
   1438     # can never be accepted, whatever its chunk alignment.
   1439     var guard = CleanupGuard()
   1440     var report = run_bounded_call(
   1441         "mutate_duplicate",
   1442         0,
   1443         300,
   1444         500,
   1445         guard,
   1446         BOUNDED_CORRELATION_MUTANT_DUPLICATE,
   1447     )
   1448     assert_true(not report.completed)
   1449     assert_equal(report.problem, "report_duplicate_report")
   1450     assert_true(report.cleanup_proved)
   1451     guard.assert_clean()
   1452 
   1453 
   1454 def test_bounded_call_rejects_overrunning_child_after_report() raises:
   1455     # BC02/BC03 before/after control: the old consumer reported a valid-looking
   1456     # completed call after killing a child that stayed alive past the budget.
   1457     # The repaired consumer must stop and reap it and report an expiry, not
   1458     # success.
   1459     var guard = CleanupGuard()
   1460     var report = run_bounded_call(
   1461         "mutate_delayed",
   1462         0,
   1463         300,
   1464         500,
   1465         guard,
   1466         BOUNDED_CORRELATION_MUTANT_DELAYED,
   1467     )
   1468     assert_true(not report.completed)
   1469     assert_true(report.stopped)
   1470     assert_true(report.cleanup_proved)
   1471     assert_true(report.elapsed_ms >= 400)
   1472     assert_true(report.elapsed_ms < 3000)
   1473     guard.assert_clean()
   1474 
   1475 
   1476 def test_bounded_report_grammar_rejects_invalid_fields() raises:
   1477     # BC02: the shared parent consumer validates the declared call grammar. A
   1478     # wrong correlation, an unknown/duplicate/missing field, a malformed token
   1479     # or an incompatible outcome is rejected as a bounded problem, never as a
   1480     # completed call for the declared kind.
   1481     var good = "report kind=max_local correlation=7 outcome=ok status=200"
   1482     var accepted = parse_bounded_report(good, "max_local", 7)
   1483     assert_true(accepted.ok)
   1484     assert_equal(accepted.status, 200)
   1485     var failing_kind = parse_bounded_report(
   1486         (
   1487             "report kind=jev correlation=7 outcome=fail cause=transport"
   1488             " reason=unknown_transport"
   1489         ),
   1490         "jev",
   1491         7,
   1492     )
   1493     assert_true(failing_kind.domain_failure())
   1494     assert_equal(failing_kind.cause, "transport")
   1495     var cases = List[String]()
   1496     cases.append(
   1497         "not_a_report kind=max_local correlation=7 outcome=ok status=200"
   1498     )
   1499     cases.append("report kind=max_local correlation=7 outcome=ok")
   1500     cases.append("report kind=max_local outcome=ok status=200")
   1501     cases.append(
   1502         "report kind=max_local correlation=7 outcome=ok status=200 bogus=1"
   1503     )
   1504     cases.append(
   1505         "report kind=max_local correlation=7 outcome=ok status=200 status=200"
   1506     )
   1507     cases.append("report kind=jev correlation=7 outcome=ok status=200")
   1508     cases.append("report kind=max_local correlation=8 outcome=ok status=200")
   1509     cases.append("report kind=max_local correlation=7 outcome=maybe status=200")
   1510     cases.append(
   1511         "report kind=max_local correlation=7 outcome=fail cause=transport "
   1512         "reason=unknown_transport status=200"
   1513     )
   1514     cases.append("report kind=max_local correlation=7 outcome=ok status=abc")
   1515     cases.append("report kind=max_local correlation=7 outcome=ok status=999")
   1516     cases.append(
   1517         "report kind=max_local correlation=7 outcome=fail cause=transport"
   1518     )
   1519     cases.append("report x")
   1520     var expected = List[String]()
   1521     expected.append("report_prefix")
   1522     expected.append("report_status_missing")
   1523     expected.append("report_missing_field")
   1524     expected.append("report_unknown_field_bogus")
   1525     expected.append("report_duplicate_field_status")
   1526     expected.append("report_kind_mismatch")
   1527     expected.append("report_correlation_mismatch")
   1528     expected.append("report_outcome_unknown_maybe")
   1529     expected.append("report_incompatible_outcome")
   1530     expected.append("report_non_numeric_status")
   1531     expected.append("report_status_invalid")
   1532     expected.append("report_cause_missing")
   1533     expected.append("report_token_grammar")
   1534     for index in range(len(cases)):
   1535         var parsed = parse_bounded_report(cases[index], "max_local", 7)
   1536         assert_true(not parsed.ok)
   1537         assert_equal(parsed.problem, expected[index])
   1538 
   1539 
   1540 def test_bounded_call_rejects_report_past_the_byte_cap() raises:
   1541     # BC02: a report line past the bounded cap is a cap-overflow harness
   1542     # failure, never a completed call.
   1543     var guard = CleanupGuard()
   1544     var report = run_bounded_call(
   1545         "mutate_huge",
   1546         0,
   1547         300,
   1548         500,
   1549         guard,
   1550         BOUNDED_CORRELATION_MUTANT_HUGE,
   1551     )
   1552     assert_true(not report.completed)
   1553     assert_equal(report.problem, "ready_output_overflow")
   1554     assert_true(report.cleanup_proved)
   1555     guard.assert_clean()
   1556 
   1557 
   1558 def test_bounded_call_transient_wait_error_is_not_success() raises:
   1559     # BC02/BC03: a transient child-exit wait error is a bounded harness failure
   1560     # that stays retryable; it can never be reported as a completed call, and
   1561     # the following cleanup still proves the exact owned child is collected.
   1562     # The wait fault is a labelled test-only seam (never a real owned child
   1563     # replaced by a synthetic identity).
   1564     var guard = CleanupGuard()
   1565     var report = run_bounded_call(
   1566         "mutate_valid",
   1567         0,
   1568         300,
   1569         500,
   1570         guard,
   1571         BOUNDED_CORRELATION_WAIT_ERROR,
   1572         "/v1/chat/completions",
   1573         "",
   1574         0,
   1575         1,
   1576     )
   1577     assert_true(not report.completed)
   1578     assert_equal(report.problem, "child_wait_error")
   1579     assert_true(report.cleanup_proved)
   1580     assert_true(report.child_status.find("exited=0") >= 0)
   1581     guard.assert_clean()
   1582 
   1583 
   1584 def test_bounded_call_rejects_early_eof_report() raises:
   1585     # RP02: an early EOF with no report byte is a harness failure, never an
   1586     # empty completed call, through the actual bounded consumer.
   1587     var guard = CleanupGuard()
   1588     var fd_before = open_fd_count_checked()
   1589     var report = run_bounded_call(
   1590         "mutate_silent",
   1591         0,
   1592         300,
   1593         500,
   1594         guard,
   1595         BOUNDED_CORRELATION_EARLY_EOF,
   1596     )
   1597     assert_true(not report.completed)
   1598     assert_equal(report.problem, "report_early_eof")
   1599     assert_true(report.cleanup_proved)
   1600     guard.assert_clean()
   1601     assert_equal(open_fd_count_checked(), fd_before)
   1602 
   1603 
   1604 def test_bounded_call_rejects_signaled_child_after_valid_report() raises:
   1605     # RP02: a valid report followed by a signaled child is a harness failure.
   1606     # The parent must observe the signal, not accept the report or kill the
   1607     # child and call it complete.
   1608     var guard = CleanupGuard()
   1609     var fd_before = open_fd_count_checked()
   1610     var report = run_bounded_call(
   1611         "mutate_signaled",
   1612         0,
   1613         300,
   1614         500,
   1615         guard,
   1616         BOUNDED_CORRELATION_SIGNALED,
   1617     )
   1618     assert_true(not report.completed)
   1619     assert_true(not report.stopped)
   1620     assert_equal(report.problem, "child_signal_9")
   1621     assert_true(report.child_status.find("signal=9") >= 0)
   1622     assert_true(report.cleanup_proved)
   1623     guard.assert_clean()
   1624     assert_equal(open_fd_count_checked(), fd_before)
   1625 
   1626 
   1627 def test_bounded_call_rejects_invalid_utf8_report() raises:
   1628     # RP02: an invalid UTF-8 byte in the report line is rejected by the bounded
   1629     # decode, so a corrupted report can never be accepted as a completed call.
   1630     var guard = CleanupGuard()
   1631     var fd_before = open_fd_count_checked()
   1632     var report = run_bounded_call(
   1633         "mutate_invalid_utf8",
   1634         0,
   1635         300,
   1636         500,
   1637         guard,
   1638         BOUNDED_CORRELATION_INVALID_UTF8,
   1639     )
   1640     assert_true(not report.completed)
   1641     assert_equal(report.problem, "invalid_utf8")
   1642     assert_true(report.cleanup_proved)
   1643     guard.assert_clean()
   1644     assert_equal(open_fd_count_checked(), fd_before)
   1645 
   1646 
   1647 def test_bounded_call_rejects_phase_isolated_late_exit() raises:
   1648     # RP02: after the report pipe is closed with one complete report, a child
   1649     # that outlives the budget must be isolated in the wait phase and stopped
   1650     # and reaped, never reported as completed.
   1651     var guard = CleanupGuard()
   1652     var fd_before = open_fd_count_checked()
   1653     var report = run_bounded_call(
   1654         "mutate_late_exit",
   1655         0,
   1656         300,
   1657         500,
   1658         guard,
   1659         BOUNDED_CORRELATION_LATE_EXIT,
   1660     )
   1661     assert_true(not report.completed)
   1662     assert_true(report.stopped)
   1663     assert_true(report.cleanup_proved)
   1664     assert_true(report.elapsed_ms >= 400)
   1665     assert_true(report.elapsed_ms < 3000)
   1666     guard.assert_clean()
   1667     assert_equal(open_fd_count_checked(), fd_before)
   1668 
   1669 
   1670 def test_bounded_call_retries_synthetic_eintr_without_error() raises:
   1671     # RP02: a bounded test-only EINTR seam exercises the actual poll consumer's
   1672     # retry branch. One interrupted poll must be retried, not reported as a read
   1673     # error, and the complete report must still be accepted. The seam is
   1674     # synthetic and disclosed as such; it changes no host signal state.
   1675     var guard = CleanupGuard()
   1676     var report = run_bounded_call(
   1677         "mutate_valid",
   1678         0,
   1679         300,
   1680         500,
   1681         guard,
   1682         BOUNDED_CORRELATION_EINTR_RETRY,
   1683         "/v1/chat/completions",
   1684         "",
   1685         0,
   1686         0,
   1687         1,
   1688         0,
   1689     )
   1690     assert_true(report.completed)
   1691     assert_equal(report.outcome, "ok")
   1692     assert_equal(report.status, 200)
   1693     assert_equal(report.problem, "")
   1694     assert_true(report.cleanup_proved)
   1695     guard.assert_clean()
   1696 
   1697 
   1698 def test_bounded_call_eintr_retry_is_bounded_by_deadline() raises:
   1699     # RP02: an unbounded synthetic EINTR storm must terminate at the caller's
   1700     # absolute deadline, never refresh the budget and never spin. The actual
   1701     # consumer reports the expiry, so the parent stops and reaps the child.
   1702     var guard = CleanupGuard()
   1703     var report = run_bounded_call(
   1704         "mutate_valid",
   1705         0,
   1706         300,
   1707         500,
   1708         guard,
   1709         BOUNDED_CORRELATION_EINTR_DEADLINE,
   1710         "/v1/chat/completions",
   1711         "",
   1712         0,
   1713         0,
   1714         -1,
   1715         0,
   1716     )
   1717     assert_true(not report.completed)
   1718     assert_true(report.stopped)
   1719     assert_true(report.cleanup_proved)
   1720     assert_true(report.elapsed_ms >= 400)
   1721     assert_true(report.elapsed_ms < 3000)
   1722     guard.assert_clean()
   1723 
   1724 
   1725 def test_bounded_call_poll_error_is_bounded_failure() raises:
   1726     # RP02: a real poll error (bounded test-only seam) is surfaced as
   1727     # read_error by the actual consumer, and the exception path still proves
   1728     # owned cleanup with no child or descriptor growth.
   1729     var guard = CleanupGuard()
   1730     var fd_before = open_fd_count_checked()
   1731     var report = run_bounded_call(
   1732         "mutate_valid",
   1733         0,
   1734         300,
   1735         500,
   1736         guard,
   1737         BOUNDED_CORRELATION_POLL_ERROR,
   1738         "/v1/chat/completions",
   1739         "",
   1740         0,
   1741         0,
   1742         0,
   1743         1,
   1744     )
   1745     assert_true(not report.completed)
   1746     assert_equal(report.problem, "read_error")
   1747     assert_true(report.cleanup_proved)
   1748     guard.assert_clean()
   1749     assert_equal(guard.pending(), 0)
   1750     assert_equal(open_fd_count_checked(), fd_before)
   1751 
   1752 
   1753 def test_maxlocal_wire_attempt_counts_are_exact() raises:
   1754     # H008: a retryable non-2xx does not trigger a hidden retry. Each explicit
   1755     # call makes exactly one counted wire attempt (request and connection).
   1756     var scripts = List[ExchangeScript]()
   1757     scripts.append(
   1758         exchange_script("ml_500", "POST", "/v1/chat/completions", 500, "{}")
   1759     )
   1760     scripts.append(
   1761         exchange_script(
   1762             "ml_ok", "POST", "/v1/chat/completions", 200, '{"choices":[]}'
   1763         )
   1764     )
   1765     var guard = CleanupGuard()
   1766     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   1767         var config = _bounded_timeout_provider_config(provider_stub.port, 3000)
   1768         var context = default_request_context()
   1769         var body = build_query_rewrite_request_body(
   1770             config, "eggs near me", context
   1771         )
   1772         var first = post_max_local_chat_completion(config, body)
   1773         assert_true(first.failure)
   1774         assert_equal(first.failure.value().kind, "http_status")
   1775         var second = post_max_local_chat_completion(config, body)
   1776         assert_true(not second.failure)
   1777         assert_equal(second.response.value().status, 200)
   1778         provider_stub.wait()
   1779         assert_true(provider_stub.ok())
   1780         assert_equal(provider_stub.request_count(), 2)
   1781         assert_equal(provider_stub.connection_count(), 2)
   1782     guard.assert_clean()
   1783 
   1784 
   1785 def test_maxlocal_wire_attempt_counts_for_non_retryable_status() raises:
   1786     # H008: a non-2xx status that must not be retried is exactly one attempt.
   1787     var scripts = List[ExchangeScript]()
   1788     scripts.append(
   1789         exchange_script("ml_400", "POST", "/v1/chat/completions", 400, "{}")
   1790     )
   1791     var guard = CleanupGuard()
   1792     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   1793         var config = _bounded_timeout_provider_config(provider_stub.port, 3000)
   1794         var context = default_request_context()
   1795         var body = build_query_rewrite_request_body(
   1796             config, "eggs near me", context
   1797         )
   1798         var outcome = post_max_local_chat_completion(config, body)
   1799         assert_true(outcome.failure)
   1800         assert_equal(outcome.failure.value().kind, "http_status")
   1801         provider_stub.wait()
   1802         assert_true(provider_stub.ok())
   1803         assert_equal(provider_stub.request_count(), 1)
   1804         assert_equal(provider_stub.connection_count(), 1)
   1805     guard.assert_clean()
   1806 
   1807 
   1808 # ── H009 query-analysis JSON boundary characterization ──────────────────────
   1809 
   1810 
   1811 def _analysis_envelope(content: String) raises -> Value:
   1812     return loads(
   1813         '{"choices":[{"message":{"content":' + json_escape(content) + "}}]}"
   1814     )
   1815 
   1816 
   1817 def _analysis_content(fields: String) -> String:
   1818     return "{" + fields + "}"
   1819 
   1820 
   1821 comptime ANALYSIS_TAIL = (
   1822     '"normalization_signals":[],"ranking_hints":[],'
   1823     '"extracted_filters":{"local_intent":true,"fulfillment":"unspecified",'
   1824     '"time_window":"unspecified"}'
   1825 )
   1826 
   1827 
   1828 def test_query_analysis_boundary_characterizes_duplicate_and_null_fields() raises:
   1829     # H009: pin current query-analysis boundary behavior. A duplicated field is
   1830     # accepted first-wins; a null string field and a non-array field are
   1831     # rejected as provider_schema_invalid.
   1832     var duplicate = _analysis_envelope(
   1833         _analysis_content(
   1834             '"original_text":"a","original_text":"b","normalized_text":"a",'
   1835             '"rewritten_text":"a","query_terms":["a"],'
   1836             + ANALYSIS_TAIL
   1837         )
   1838     )
   1839     var analysis = parse_query_analysis_from_chat_completion(duplicate)
   1840     assert_equal(analysis.original_text, "a")
   1841     assert_equal(len(analysis.query_terms), 1)
   1842     var null_text = _analysis_envelope(
   1843         _analysis_content(
   1844             '"original_text":null,"normalized_text":"a","rewritten_text":"a",'
   1845             '"query_terms":["a"],'
   1846             + ANALYSIS_TAIL
   1847         )
   1848     )
   1849     var message = ""
   1850     try:
   1851         _ = parse_query_analysis_from_chat_completion(null_text)
   1852     except e:
   1853         message = String(e)
   1854     assert_true(message.find("provider_schema_invalid") >= 0)
   1855     var numeric_terms = _analysis_envelope(
   1856         _analysis_content(
   1857             '"original_text":"a","normalized_text":"a","rewritten_text":"a",'
   1858             '"query_terms":7,'
   1859             + ANALYSIS_TAIL
   1860         )
   1861     )
   1862     var terms_message = ""
   1863     try:
   1864         _ = parse_query_analysis_from_chat_completion(numeric_terms)
   1865     except e:
   1866         terms_message = String(e)
   1867     assert_true(terms_message.find("provider_schema_invalid") >= 0)
   1868 
   1869 
   1870 def _raw_bytes(text: String) -> List[UInt8]:
   1871     var raw = List[UInt8]()
   1872     for byte in text.as_bytes():
   1873         raw.append(UInt8(Int(byte)))
   1874     return raw^
   1875 
   1876 
   1877 def _repeat_text(mark: String, count: Int) -> String:
   1878     var out = List[UInt8]()
   1879     var mark_bytes = mark.as_bytes()
   1880     for _ in range(count):
   1881         for index in range(len(mark_bytes)):
   1882             out.append(UInt8(Int(mark_bytes[index])))
   1883     return String(unsafe_from_utf8=Span(ptr=out.unsafe_ptr(), length=len(out)))
   1884 
   1885 
   1886 # OB01: the exact raw response grammar. A status-looking value inside a header
   1887 # is never the status; a different field name is preserved as unknown; junk,
   1888 # duplicate, conflicting, negative, overflowing or unsupported framing is an
   1889 # explicit bounded problem rather than a silently absent field.
   1890 
   1891 
   1892 def test_raw_accounting_preserves_unknown_length_like_field() raises:
   1893     var accounting = account_raw_bytes(
   1894         _raw_bytes("HTTP/1.1 200 OK\r\nX-Content-Length: 3\r\n\r\nabc"),
   1895         "abc",
   1896     )
   1897     assert_equal(accounting.problem, "")
   1898     assert_equal(accounting.status, 200)
   1899     assert_equal(accounting.body_bytes, 3)
   1900     assert_equal(accounting.declared_bytes, -1)
   1901     assert_equal(accounting.length_match, "unknown")
   1902     assert_equal(accounting.surplus_bytes, 0)
   1903 
   1904 
   1905 def test_raw_accounting_rejects_junk_content_length() raises:
   1906     var accounting = account_raw_bytes(
   1907         _raw_bytes("HTTP/1.1 200 OK\r\nContent-Length: 3junk\r\n\r\nabc"),
   1908         "abc",
   1909     )
   1910     assert_equal(accounting.problem, "raw_content_length_malformed")
   1911 
   1912 
   1913 def test_raw_accounting_rejects_identical_duplicate_length() raises:
   1914     var accounting = account_raw_bytes(
   1915         _raw_bytes(
   1916             "HTTP/1.1 200 OK\r\nContent-Length: 3\r\n"
   1917             "Content-Length: 3\r\n\r\nabc"
   1918         ),
   1919         "abc",
   1920     )
   1921     assert_equal(accounting.problem, "raw_content_length_duplicate")
   1922 
   1923 
   1924 def test_raw_accounting_rejects_conflicting_duplicate_length() raises:
   1925     var accounting = account_raw_bytes(
   1926         _raw_bytes(
   1927             "HTTP/1.1 200 OK\r\nContent-Length: 3\r\n"
   1928             "Content-Length: 9\r\n\r\nabc"
   1929         ),
   1930         "abc",
   1931     )
   1932     assert_equal(accounting.problem, "raw_content_length_duplicate")
   1933 
   1934 
   1935 def test_raw_accounting_rejects_garbage_status_with_header_status_value() raises:
   1936     # A status-looking string in a header after a garbage first line must never
   1937     # be read as the response status.
   1938     var accounting = account_raw_bytes(
   1939         _raw_bytes(
   1940             "garbage\r\nx-note: HTTP/1.1 200 OK\r\nContent-Length: 3\r\n\r\nabc"
   1941         ),
   1942         "abc",
   1943     )
   1944     assert_equal(accounting.problem, "raw_status_missing")
   1945     assert_equal(accounting.status, 0)
   1946 
   1947 
   1948 def test_raw_accounting_rejects_unsupported_transfer_encoding() raises:
   1949     var accounting = account_raw_bytes(
   1950         _raw_bytes("HTTP/1.1 200 OK\r\ntransfer-encoding: chunked\r\n\r\nabc"),
   1951         "abc",
   1952     )
   1953     assert_equal(accounting.problem, "raw_transfer_encoding_unsupported")
   1954 
   1955 
   1956 def test_raw_accounting_rejects_negative_content_length() raises:
   1957     var accounting = account_raw_bytes(
   1958         _raw_bytes("HTTP/1.1 200 OK\r\nContent-Length: -3\r\n\r\nabc"),
   1959         "abc",
   1960     )
   1961     assert_equal(accounting.problem, "raw_content_length_malformed")
   1962 
   1963 
   1964 def test_raw_accounting_rejects_overflow_content_length() raises:
   1965     var accounting = account_raw_bytes(
   1966         _raw_bytes(
   1967             "HTTP/1.1 200 OK\r\nContent-Length: 99999999999999999999\r\n\r\nabc"
   1968         ),
   1969         "abc",
   1970     )
   1971     assert_equal(accounting.problem, "raw_content_length_overflow")
   1972 
   1973 
   1974 def test_raw_accounting_rejects_malformed_header_line() raises:
   1975     var accounting = account_raw_bytes(
   1976         _raw_bytes("HTTP/1.1 200 OK\r\nbad header line\r\n\r\nabc"), "abc"
   1977     )
   1978     assert_equal(accounting.problem, "raw_header_malformed")
   1979 
   1980 
   1981 def test_raw_accounting_rejects_out_of_range_status() raises:
   1982     # OB01: only a valid three-digit HTTP status (100..599) is accepted.
   1983     var low = account_raw_bytes(
   1984         _raw_bytes("HTTP/1.1 000 X\r\ncontent-length: 3\r\n\r\nabc"), "abc"
   1985     )
   1986     assert_equal(low.problem, "raw_status_malformed")
   1987     var high = account_raw_bytes(
   1988         _raw_bytes("HTTP/1.1 999 X\r\ncontent-length: 3\r\n\r\nabc"), "abc"
   1989     )
   1990     assert_equal(high.problem, "raw_status_malformed")
   1991 
   1992 
   1993 def test_raw_accounting_accepts_identity_transfer_encoding() raises:
   1994     var accounting = account_raw_bytes(
   1995         _raw_bytes("HTTP/1.1 200 OK\r\ntransfer-encoding: identity\r\n\r\nabc"),
   1996         "abc",
   1997     )
   1998     assert_equal(accounting.problem, "")
   1999     assert_equal(accounting.status, 200)
   2000     assert_equal(accounting.declared_bytes, -1)
   2001     assert_equal(accounting.length_match, "unknown")
   2002 
   2003 
   2004 def test_raw_accounting_rejects_transfer_with_length_conflict() raises:
   2005     var accounting = account_raw_bytes(
   2006         _raw_bytes(
   2007             "HTTP/1.1 200 OK\r\ntransfer-encoding: identity\r\n"
   2008             "content-length: 3\r\n\r\nabc"
   2009         ),
   2010         "abc",
   2011     )
   2012     assert_equal(accounting.problem, "raw_framing_conflict")
   2013 
   2014 
   2015 def test_raw_accounting_reports_buffered_surplus() raises:
   2016     # OB02: bytes already buffered beyond the declared body are reported
   2017     # explicitly instead of being silently folded into the declared length.
   2018     var accounting = account_raw_bytes(
   2019         _raw_bytes("HTTP/1.1 200 OK\r\ncontent-length: 3\r\n\r\nabcde"),
   2020         "abc",
   2021     )
   2022     assert_equal(accounting.problem, "")
   2023     assert_equal(accounting.body_bytes, 5)
   2024     assert_equal(accounting.declared_bytes, 3)
   2025     assert_equal(accounting.length_match, "no")
   2026     assert_equal(accounting.surplus_bytes, 2)
   2027 
   2028 
   2029 # OB02 acquisition controls through the actual incremental read path.
   2030 
   2031 
   2032 def test_provider_raw_head_body_accounts_buffered_surplus() raises:
   2033     var scripts = List[ExchangeScript]()
   2034     var script = exchange_script(
   2035         "buffered_surplus", "POST", "/v1/chat/completions", 200, ""
   2036     )
   2037     script.raw_response = (
   2038         "HTTP/1.1 200 OK\r\ncontent-length: 3\r\nconnection: close\r\n\r\nabcde"
   2039     )
   2040     scripts.append(script^)
   2041     var guard = CleanupGuard()
   2042     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   2043         var report = run_bounded_call(
   2044             "raw_head_body",
   2045             provider_stub.port,
   2046             5000,
   2047             5000,
   2048             guard,
   2049             BOUNDED_CORRELATION_OB_SURPLUS,
   2050             "/v1/chat/completions",
   2051             "abc",
   2052         )
   2053         assert_true(report.ok())
   2054         assert_equal(report.status, 200)
   2055         assert_equal(report.body_bytes, 5)
   2056         assert_equal(report.declared_bytes, 3)
   2057         assert_equal(report.length_match, "no")
   2058         assert_equal(report.surplus_bytes, 2)
   2059         provider_stub.wait()
   2060         assert_true(provider_stub.ok())
   2061     guard.assert_clean()
   2062 
   2063 
   2064 def test_provider_raw_head_body_rejects_declared_over_cap() raises:
   2065     # OB02: a declared length beyond the body cap is an explicit overflow, never
   2066     # an accepted observation with a silently truncated prefix.
   2067     var scripts = List[ExchangeScript]()
   2068     var script = exchange_script(
   2069         "declared_over_cap", "POST", "/v1/chat/completions", 200, ""
   2070     )
   2071     script.raw_response = (
   2072         "HTTP/1.1 200 OK\r\ncontent-length: 99999999\r\n"
   2073         "connection: close\r\n\r\nabc"
   2074     )
   2075     scripts.append(script^)
   2076     var guard = CleanupGuard()
   2077     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   2078         var report = run_bounded_call(
   2079             "raw_head_body",
   2080             provider_stub.port,
   2081             5000,
   2082             5000,
   2083             guard,
   2084             BOUNDED_CORRELATION_OB_OVER_CAP,
   2085             "/v1/chat/completions",
   2086             "abc",
   2087         )
   2088         assert_true(report.completed)
   2089         assert_true(report.domain_failure())
   2090         assert_equal(report.cause, "raw_body_overflow")
   2091         provider_stub.wait()
   2092         assert_true(provider_stub.ok())
   2093     guard.assert_clean()
   2094 
   2095 
   2096 def test_provider_raw_head_body_rejects_duplicate_length() raises:
   2097     var scripts = List[ExchangeScript]()
   2098     var script = exchange_script(
   2099         "duplicate_length", "POST", "/v1/chat/completions", 200, ""
   2100     )
   2101     script.raw_response = (
   2102         "HTTP/1.1 200 OK\r\ncontent-length: 3\r\ncontent-length: 3\r\n"
   2103         "connection: close\r\n\r\nabc"
   2104     )
   2105     scripts.append(script^)
   2106     var guard = CleanupGuard()
   2107     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   2108         var report = run_bounded_call(
   2109             "raw_head_body",
   2110             provider_stub.port,
   2111             5000,
   2112             5000,
   2113             guard,
   2114             BOUNDED_CORRELATION_OB_DUP_LENGTH,
   2115             "/v1/chat/completions",
   2116             "abc",
   2117         )
   2118         assert_true(report.completed)
   2119         assert_true(report.domain_failure())
   2120         assert_equal(report.cause, "raw_content_length_duplicate")
   2121         provider_stub.wait()
   2122         assert_true(provider_stub.ok())
   2123     guard.assert_clean()
   2124 
   2125 
   2126 def test_provider_raw_head_body_no_length_completes_at_eof() raises:
   2127     # OB02: a response without a declared length completes at EOF inside the
   2128     # cap, with an explicitly unknown length rather than an invented one.
   2129     var scripts = List[ExchangeScript]()
   2130     var script = exchange_script(
   2131         "no_length", "POST", "/v1/chat/completions", 200, ""
   2132     )
   2133     script.raw_response = (
   2134         "HTTP/1.1 200 OK\r\nconnection: close\r\n\r\nNOLENGTHBODY"
   2135     )
   2136     scripts.append(script^)
   2137     var guard = CleanupGuard()
   2138     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   2139         var report = run_bounded_call(
   2140             "raw_head_body",
   2141             provider_stub.port,
   2142             5000,
   2143             5000,
   2144             guard,
   2145             BOUNDED_CORRELATION_OB_NO_LENGTH,
   2146             "/v1/chat/completions",
   2147             "NOLENGTHBODY",
   2148         )
   2149         assert_true(report.ok())
   2150         assert_equal(report.status, 200)
   2151         assert_equal(report.body_bytes, "NOLENGTHBODY".byte_length())
   2152         assert_equal(report.body_match, "yes")
   2153         assert_equal(report.declared_bytes, -1)
   2154         assert_equal(report.length_match, "unknown")
   2155         assert_equal(report.surplus_bytes, 0)
   2156         provider_stub.wait()
   2157         assert_true(provider_stub.ok())
   2158     guard.assert_clean()
   2159 
   2160 
   2161 def test_provider_raw_head_body_split_header_terminator_is_preserved() raises:
   2162     # OB02: the real incremental acquisition path is driven with one-byte reads,
   2163     # so the CRLFCRLF terminator is split across reads. Concatenating loops
   2164     # before a pure parser call is not fragmentation proof, so the seam caps the
   2165     # actual read.
   2166     var scripts = List[ExchangeScript]()
   2167     scripts.append(
   2168         exchange_script(
   2169             "split_terminator", "POST", "/v1/chat/completions", 200, "BODYMARK"
   2170         )
   2171     )
   2172     var plan = List[Int]()
   2173     plan.append(1)
   2174     var guard = CleanupGuard()
   2175     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   2176         var report = run_bounded_call(
   2177             "raw_head_body",
   2178             provider_stub.port,
   2179             5000,
   2180             5000,
   2181             guard,
   2182             BOUNDED_CORRELATION_OB_SPLIT_TERMINATOR,
   2183             "/v1/chat/completions",
   2184             "BODYMARK",
   2185             raw_chunk_plan=plan,
   2186         )
   2187         assert_true(report.ok())
   2188         assert_equal(report.status, 200)
   2189         assert_equal(report.body_bytes, 8)
   2190         assert_equal(report.body_match, "yes")
   2191         assert_equal(report.declared_bytes, 8)
   2192         assert_equal(report.length_match, "yes")
   2193         assert_equal(report.surplus_bytes, 0)
   2194         provider_stub.wait()
   2195         assert_true(provider_stub.ok())
   2196     guard.assert_clean()
   2197 
   2198 
   2199 def test_provider_raw_head_body_split_multibyte_body_is_preserved() raises:
   2200     # OB02: a one-byte read seam splits the three-byte UTF-8 character across
   2201     # reads; the byte accumulation must still preserve it whole.
   2202     var scripts = List[ExchangeScript]()
   2203     scripts.append(
   2204         exchange_script(
   2205             "split_multibyte",
   2206             "POST",
   2207             "/v1/chat/completions",
   2208             200,
   2209             '{"mark":"a☃b"}',
   2210         )
   2211     )
   2212     var plan = List[Int]()
   2213     plan.append(1)
   2214     var guard = CleanupGuard()
   2215     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   2216         var report = run_bounded_call(
   2217             "raw_head_body",
   2218             provider_stub.port,
   2219             5000,
   2220             5000,
   2221             guard,
   2222             BOUNDED_CORRELATION_OB_SPLIT_BODY,
   2223             "/v1/chat/completions",
   2224             "a☃b",
   2225             raw_chunk_plan=plan,
   2226         )
   2227         assert_true(report.ok())
   2228         assert_equal(report.status, 200)
   2229         assert_equal(report.body_bytes, '{"mark":"a☃b"}'.byte_length())
   2230         assert_equal(report.body_match, "yes")
   2231         assert_equal(report.length_match, "yes")
   2232         provider_stub.wait()
   2233         assert_true(provider_stub.ok())
   2234     guard.assert_clean()
   2235 
   2236 
   2237 def test_provider_raw_head_body_near_cap_coalesced_body_is_not_head_overflow() raises:
   2238     # OB02: the header cap counts bytes through the terminator only. A valid
   2239     # near-cap header whose final 1000-byte read also carries coalesced body
   2240     # bytes must not be reported as a header overflow.
   2241     var pad = _repeat_text("a", 65459)
   2242     var body = "ENDMARK" + _repeat_text("y", 593)
   2243     var raw = (
   2244         "HTTP/1.1 200 OK\r\nx-pad: "
   2245         + pad
   2246         + "\r\ncontent-length: 600\r\nconnection: close\r\n\r\n"
   2247         + body
   2248     )
   2249     var scripts = List[ExchangeScript]()
   2250     var script = exchange_script(
   2251         "near_cap", "POST", "/v1/chat/completions", 200, ""
   2252     )
   2253     script.raw_response = raw
   2254     scripts.append(script^)
   2255     var plan = List[Int]()
   2256     plan.append(1000)
   2257     var guard = CleanupGuard()
   2258     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   2259         var report = run_bounded_call(
   2260             "raw_head_body",
   2261             provider_stub.port,
   2262             5000,
   2263             5000,
   2264             guard,
   2265             BOUNDED_CORRELATION_OB_NEAR_CAP,
   2266             "/v1/chat/completions",
   2267             "ENDMARK",
   2268             raw_chunk_plan=plan,
   2269         )
   2270         assert_true(report.ok())
   2271         assert_equal(report.status, 200)
   2272         assert_equal(report.body_bytes, 600)
   2273         assert_equal(report.body_match, "yes")
   2274         assert_equal(report.declared_bytes, 600)
   2275         assert_equal(report.length_match, "yes")
   2276         assert_equal(report.surplus_bytes, 0)
   2277         provider_stub.wait()
   2278         assert_true(provider_stub.ok())
   2279     guard.assert_clean()
   2280 
   2281 
   2282 def test_provider_raw_head_body_reports_malformed_length_grammar() raises:
   2283     var scripts = List[ExchangeScript]()
   2284     var script = exchange_script(
   2285         "malformed_length", "POST", "/v1/chat/completions", 200, ""
   2286     )
   2287     script.raw_response = (
   2288         "HTTP/1.1 200 OK\r\ncontent-length: 3junk\r\nconnection:"
   2289         " close\r\n\r\nabc"
   2290     )
   2291     scripts.append(script^)
   2292     var guard = CleanupGuard()
   2293     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   2294         var report = run_bounded_call(
   2295             "raw_head_body",
   2296             provider_stub.port,
   2297             5000,
   2298             5000,
   2299             guard,
   2300             BOUNDED_CORRELATION_OB_MALFORMED_HEAD,
   2301             "/v1/chat/completions",
   2302             "abc",
   2303         )
   2304         assert_true(report.completed)
   2305         assert_true(report.domain_failure())
   2306         assert_equal(report.cause, "raw_content_length_malformed")
   2307         provider_stub.wait()
   2308         assert_true(provider_stub.ok())
   2309     guard.assert_clean()
   2310 
   2311 
   2312 # D44/IL01: the inclusive 1,048,576-byte raw-body cap, proven through the
   2313 # actual incremental acquisition path for both a declared Content-Length and an
   2314 # EOF-delimited (no-length) response. Exact cap is inclusive; a real extra byte
   2315 # fails overflow; a stalled peer fails the existing parent deadline (timeout),
   2316 # never an invented overflow.
   2317 
   2318 comptime RAW_BODY_CAP_TEST = 1048576
   2319 
   2320 
   2321 def _cap_body(count: Int) -> String:
   2322     return _repeat_text("a", count)
   2323 
   2324 
   2325 def test_provider_raw_head_body_declared_cap_minus_one_succeeds() raises:
   2326     var body = _cap_body(RAW_BODY_CAP_TEST - 1)
   2327     var scripts = List[ExchangeScript]()
   2328     var script = exchange_script(
   2329         "declared_cap_m1", "POST", "/v1/chat/completions", 200, ""
   2330     )
   2331     script.raw_response = (
   2332         "HTTP/1.1 200 OK\r\ncontent-length: "
   2333         + String(RAW_BODY_CAP_TEST - 1)
   2334         + "\r\nconnection: close\r\n\r\n"
   2335         + body
   2336     )
   2337     scripts.append(script^)
   2338     var guard = CleanupGuard()
   2339     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   2340         var report = run_bounded_call(
   2341             "raw_head_body",
   2342             provider_stub.port,
   2343             10000,
   2344             10000,
   2345             guard,
   2346             BOUNDED_CORRELATION_OB_CAP_DECLARED_MINUS,
   2347             "/v1/chat/completions",
   2348             "aaa",
   2349         )
   2350         assert_true(report.ok())
   2351         assert_equal(report.status, 200)
   2352         assert_equal(report.body_bytes, RAW_BODY_CAP_TEST - 1)
   2353         assert_equal(report.declared_bytes, RAW_BODY_CAP_TEST - 1)
   2354         assert_equal(report.length_match, "yes")
   2355         assert_equal(report.surplus_bytes, 0)
   2356         provider_stub.wait()
   2357         assert_true(provider_stub.ok())
   2358     guard.assert_clean()
   2359 
   2360 
   2361 def test_provider_raw_head_body_declared_cap_exact_succeeds() raises:
   2362     # D44/IL01: a declared body of exactly the cap is inclusive and completes
   2363     # without a sentinel read, because the declared length already bounds it.
   2364     var body = _cap_body(RAW_BODY_CAP_TEST)
   2365     var scripts = List[ExchangeScript]()
   2366     var script = exchange_script(
   2367         "declared_cap_exact", "POST", "/v1/chat/completions", 200, ""
   2368     )
   2369     script.raw_response = (
   2370         "HTTP/1.1 200 OK\r\ncontent-length: "
   2371         + String(RAW_BODY_CAP_TEST)
   2372         + "\r\nconnection: close\r\n\r\n"
   2373         + body
   2374     )
   2375     scripts.append(script^)
   2376     var guard = CleanupGuard()
   2377     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   2378         var report = run_bounded_call(
   2379             "raw_head_body",
   2380             provider_stub.port,
   2381             10000,
   2382             10000,
   2383             guard,
   2384             BOUNDED_CORRELATION_OB_CAP_DECLARED_EXACT,
   2385             "/v1/chat/completions",
   2386             "aaa",
   2387         )
   2388         assert_true(report.ok())
   2389         assert_equal(report.status, 200)
   2390         assert_equal(report.body_bytes, RAW_BODY_CAP_TEST)
   2391         assert_equal(report.declared_bytes, RAW_BODY_CAP_TEST)
   2392         assert_equal(report.length_match, "yes")
   2393         assert_equal(report.surplus_bytes, 0)
   2394         provider_stub.wait()
   2395         assert_true(provider_stub.ok())
   2396     guard.assert_clean()
   2397 
   2398 
   2399 def test_provider_raw_head_body_declared_cap_plus_one_is_over_cap() raises:
   2400     # D44/IL01: a declared length above the cap is rejected at the head grammar
   2401     # (declared-length rejection), distinctly from an EOF acquisition overflow.
   2402     var scripts = List[ExchangeScript]()
   2403     var script = exchange_script(
   2404         "declared_cap_p1", "POST", "/v1/chat/completions", 200, ""
   2405     )
   2406     script.raw_response = (
   2407         "HTTP/1.1 200 OK\r\ncontent-length: "
   2408         + String(RAW_BODY_CAP_TEST + 1)
   2409         + "\r\nconnection: close\r\n\r\nabc"
   2410     )
   2411     scripts.append(script^)
   2412     var guard = CleanupGuard()
   2413     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   2414         var report = run_bounded_call(
   2415             "raw_head_body",
   2416             provider_stub.port,
   2417             10000,
   2418             10000,
   2419             guard,
   2420             BOUNDED_CORRELATION_OB_CAP_DECLARED_PLUS,
   2421             "/v1/chat/completions",
   2422             "abc",
   2423         )
   2424         assert_true(report.completed)
   2425         assert_true(report.domain_failure())
   2426         assert_equal(report.cause, "raw_body_overflow")
   2427         assert_equal(report.reason, "raw_body_overflow")
   2428         provider_stub.wait()
   2429         assert_true(provider_stub.ok())
   2430     guard.assert_clean()
   2431 
   2432 
   2433 def test_provider_raw_head_body_no_length_cap_minus_one_succeeds() raises:
   2434     var body = _cap_body(RAW_BODY_CAP_TEST - 1)
   2435     var scripts = List[ExchangeScript]()
   2436     var script = exchange_script(
   2437         "nolen_cap_m1", "POST", "/v1/chat/completions", 200, ""
   2438     )
   2439     script.raw_response = "HTTP/1.1 200 OK\r\nconnection: close\r\n\r\n" + body
   2440     scripts.append(script^)
   2441     var guard = CleanupGuard()
   2442     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   2443         var report = run_bounded_call(
   2444             "raw_head_body",
   2445             provider_stub.port,
   2446             10000,
   2447             10000,
   2448             guard,
   2449             BOUNDED_CORRELATION_OB_CAP_NO_LENGTH_MINUS,
   2450             "/v1/chat/completions",
   2451             "aaa",
   2452         )
   2453         assert_true(report.ok())
   2454         assert_equal(report.status, 200)
   2455         assert_equal(report.body_bytes, RAW_BODY_CAP_TEST - 1)
   2456         assert_equal(report.declared_bytes, -1)
   2457         assert_equal(report.length_match, "unknown")
   2458         assert_equal(report.surplus_bytes, 0)
   2459         provider_stub.wait()
   2460         assert_true(provider_stub.ok())
   2461     guard.assert_clean()
   2462 
   2463 
   2464 def test_provider_raw_head_body_no_length_exact_cap_succeeds_at_eof() raises:
   2465     # D44/IL01 regression: the exact inclusive cap is a success when the peer
   2466     # closed at the cap. The one-byte sentinel observes EOF and is never
   2467     # appended, so body_bytes stays at the cap.
   2468     var body = _cap_body(RAW_BODY_CAP_TEST)
   2469     var scripts = List[ExchangeScript]()
   2470     var script = exchange_script(
   2471         "nolen_cap_exact", "POST", "/v1/chat/completions", 200, ""
   2472     )
   2473     script.raw_response = "HTTP/1.1 200 OK\r\nconnection: close\r\n\r\n" + body
   2474     scripts.append(script^)
   2475     var guard = CleanupGuard()
   2476     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   2477         var report = run_bounded_call(
   2478             "raw_head_body",
   2479             provider_stub.port,
   2480             10000,
   2481             10000,
   2482             guard,
   2483             BOUNDED_CORRELATION_OB_CAP_NO_LENGTH_EXACT,
   2484             "/v1/chat/completions",
   2485             "aaa",
   2486         )
   2487         assert_true(report.ok())
   2488         assert_equal(report.status, 200)
   2489         assert_equal(report.body_bytes, RAW_BODY_CAP_TEST)
   2490         assert_equal(report.declared_bytes, -1)
   2491         assert_equal(report.length_match, "unknown")
   2492         assert_equal(report.surplus_bytes, 0)
   2493         provider_stub.wait()
   2494         assert_true(provider_stub.ok())
   2495     guard.assert_clean()
   2496 
   2497 
   2498 def test_provider_raw_head_body_no_length_cap_plus_one_overflows() raises:
   2499     # D44/IL01: a real extra byte beyond the cap fails overflow through the
   2500     # sentinel read; the sentinel is never appended, so body_bytes stays at the
   2501     # inclusive maximum rather than exceeding it.
   2502     var body = _cap_body(RAW_BODY_CAP_TEST + 1)
   2503     var scripts = List[ExchangeScript]()
   2504     var script = exchange_script(
   2505         "nolen_cap_p1", "POST", "/v1/chat/completions", 200, ""
   2506     )
   2507     script.raw_response = "HTTP/1.1 200 OK\r\nconnection: close\r\n\r\n" + body
   2508     scripts.append(script^)
   2509     var guard = CleanupGuard()
   2510     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   2511         var report = run_bounded_call(
   2512             "raw_head_body",
   2513             provider_stub.port,
   2514             10000,
   2515             10000,
   2516             guard,
   2517             BOUNDED_CORRELATION_OB_CAP_NO_LENGTH_PLUS,
   2518             "/v1/chat/completions",
   2519             "aaa",
   2520         )
   2521         assert_true(report.completed)
   2522         assert_true(report.domain_failure())
   2523         assert_equal(report.cause, "raw_body_overflow")
   2524         assert_equal(report.reason, "body_overflow")
   2525         assert_equal(report.body_bytes, RAW_BODY_CAP_TEST)
   2526         assert_equal(report.surplus_bytes, 0)
   2527         provider_stub.wait()
   2528         assert_true(provider_stub.ok())
   2529     guard.assert_clean()
   2530 
   2531 
   2532 def test_provider_raw_head_body_no_length_exact_cap_stall_is_timeout() raises:
   2533     # D44/IL01: a peer that delivers exactly the cap and then stalls must be
   2534     # reported by the existing parent deadline as a stopped (timeout) call,
   2535     # never as an invented raw_body_overflow. The fixture keeps the connection
   2536     # open, so only the parent deadline can end the call.
   2537     var body = _cap_body(RAW_BODY_CAP_TEST)
   2538     var scripts = List[ExchangeScript]()
   2539     var script = exchange_script(
   2540         "nolen_cap_stall", "POST", "/v1/chat/completions", 200, ""
   2541     )
   2542     script.raw_response = "HTTP/1.1 200 OK\r\nconnection: close\r\n\r\n" + body
   2543     script.close_connection = False
   2544     scripts.append(script^)
   2545     scripts.append(
   2546         exchange_script(
   2547             "nolen_cap_stall_unused", "POST", "/v1/chat/completions", 200, ""
   2548         )
   2549     )
   2550     var guard = CleanupGuard()
   2551     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   2552         var report = run_bounded_call(
   2553             "raw_head_body",
   2554             provider_stub.port,
   2555             10000,
   2556             1500,
   2557             guard,
   2558             BOUNDED_CORRELATION_OB_CAP_NO_LENGTH_STALL,
   2559             "/v1/chat/completions",
   2560             "aaa",
   2561         )
   2562         assert_true(report.stopped)
   2563         assert_true(not report.completed)
   2564         assert_true(report.cleanup_proved)
   2565         assert_equal(report.problem, "")
   2566         assert_equal(report.cause, "")
   2567     guard.assert_clean()
   2568 
   2569 
   2570 # OB03: a synthetic write-error seam is never a real peer close.
   2571 
   2572 
   2573 def _assert_synthetic_errno_rejected(errno: Int, declared: Bool) raises:
   2574     var scripts = List[ExchangeScript]()
   2575     var script = exchange_script(
   2576         "synthetic_errno", "POST", "/v1/chat/completions", 200, '{"choices":[]}'
   2577     )
   2578     script.stall_after_head_ms = 200
   2579     if declared:
   2580         script.expect_peer_close = True
   2581         script.expected_close_cause = "broken_pipe"
   2582         script.expected_close_phase = "body_stall"
   2583     script.inject_write_errno = errno
   2584     scripts.append(script^)
   2585     var guard = CleanupGuard()
   2586     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   2587         _ = _raw_send_and_read(provider_stub.port, "/v1/chat/completions")
   2588         provider_stub.reap()
   2589         assert_true(not provider_stub.ok())
   2590         assert_equal(provider_stub.phase(), "peer_close")
   2591         assert_true(provider_stub.reason().find("unexpected_write_") >= 0)
   2592         assert_true(provider_stub.reason().find("synthdecl") >= 0)
   2593     guard.assert_clean()
   2594 
   2595 
   2596 def test_provider_scripted_epipe_with_declaration_is_not_peer_close() raises:
   2597     _assert_synthetic_errno_rejected(Int(ErrNo.EPIPE.value), True)
   2598 
   2599 
   2600 def test_provider_scripted_epipe_without_declaration_is_not_peer_close() raises:
   2601     _assert_synthetic_errno_rejected(Int(ErrNo.EPIPE.value), False)
   2602 
   2603 
   2604 def test_provider_scripted_econnreset_with_declaration_is_not_peer_close() raises:
   2605     _assert_synthetic_errno_rejected(Int(ErrNo.ECONNRESET.value), True)
   2606 
   2607 
   2608 def test_provider_scripted_econnreset_without_declaration_is_not_peer_close() raises:
   2609     _assert_synthetic_errno_rejected(Int(ErrNo.ECONNRESET.value), False)
   2610 
   2611 
   2612 def test_provider_scripted_eagain_with_declaration_is_not_peer_close() raises:
   2613     _assert_synthetic_errno_rejected(Int(ErrNo.EAGAIN.value), True)
   2614 
   2615 
   2616 def test_provider_scripted_eagain_without_declaration_is_not_peer_close() raises:
   2617     _assert_synthetic_errno_rejected(Int(ErrNo.EAGAIN.value), False)
   2618 
   2619 
   2620 def test_provider_scripted_ebadf_with_declaration_is_not_peer_close() raises:
   2621     _assert_synthetic_errno_rejected(Int(ErrNo.EBADF.value), True)
   2622 
   2623 
   2624 def test_provider_scripted_ebadf_without_declaration_is_not_peer_close() raises:
   2625     _assert_synthetic_errno_rejected(Int(ErrNo.EBADF.value), False)
   2626 
   2627 
   2628 def _assert_synthetic_errno_delayed_write_rejected(errno: Int) raises:
   2629     # OB03: the same provenance rule is exercised on the non-stall delayed_write
   2630     # write step, not only the body_stall step.
   2631     var scripts = List[ExchangeScript]()
   2632     var script = exchange_script(
   2633         "synthetic_delayed",
   2634         "POST",
   2635         "/v1/chat/completions",
   2636         200,
   2637         '{"choices":[]}',
   2638     )
   2639     script.expect_peer_close = True
   2640     script.expected_close_cause = "broken_pipe"
   2641     script.expected_close_phase = "delayed_write"
   2642     script.inject_write_errno = errno
   2643     scripts.append(script^)
   2644     var guard = CleanupGuard()
   2645     with spawn_max_local_scripted(0, scripts^, guard) as provider_stub:
   2646         _ = _raw_send_and_read(provider_stub.port, "/v1/chat/completions")
   2647         provider_stub.reap()
   2648         assert_true(not provider_stub.ok())
   2649         assert_equal(provider_stub.phase(), "peer_close")
   2650         assert_true(provider_stub.reason().find("unexpected_write_") >= 0)
   2651         assert_true(provider_stub.reason().find("_delayed_write_") >= 0)
   2652         assert_true(provider_stub.reason().find("synthdecl") >= 0)
   2653     guard.assert_clean()
   2654 
   2655 
   2656 def test_provider_scripted_epipe_delayed_write_is_not_peer_close() raises:
   2657     _assert_synthetic_errno_delayed_write_rejected(Int(ErrNo.EPIPE.value))
   2658 
   2659 
   2660 def test_provider_scripted_econnreset_delayed_write_is_not_peer_close() raises:
   2661     _assert_synthetic_errno_delayed_write_rejected(Int(ErrNo.ECONNRESET.value))
   2662 
   2663 
   2664 def test_provider_scripted_eagain_delayed_write_is_not_peer_close() raises:
   2665     _assert_synthetic_errno_delayed_write_rejected(Int(ErrNo.EAGAIN.value))
   2666 
   2667 
   2668 def test_provider_scripted_ebadf_delayed_write_is_not_peer_close() raises:
   2669     _assert_synthetic_errno_delayed_write_rejected(Int(ErrNo.EBADF.value))
   2670 
   2671 
   2672 def main() raises:
   2673     TestSuite.discover_tests[__functions_in_module()]().run()