#!/usr/bin/env python3
"""POST /v1/evaluate round trip against the local HTTP service.

Starts helixor_runtime.server.start_server on 127.0.0.1 (random port), uses a
single persistent keep-alive connection (http.client), 50 warm-up requests,
then 1,000 timed sequential requests cycling over five payloads.
"""
import http.client
import json
import time
from urllib.parse import urlparse

from helixor_runtime.server import start_server, stop_server

from _common import PAYLOADS, environment, summarize

WARMUP = 50
N = 1_000


def main() -> None:
    print(environment())
    server, thread, base_url = start_server(host="127.0.0.1", port=0)
    try:
        u = urlparse(base_url)
        conn = http.client.HTTPConnection(u.hostname, u.port, timeout=10)
        bodies = [json.dumps({"text": p}).encode() for p in PAYLOADS]
        headers = {"Content-Type": "application/json"}

        def call(i: int) -> dict:
            conn.request("POST", "/v1/evaluate", body=bodies[i % len(bodies)], headers=headers)
            resp = conn.getresponse()
            data = resp.read()
            if resp.status != 200:
                raise RuntimeError(f"HTTP {resp.status}: {data[:200]!r}")
            return json.loads(data)

        for i in range(WARMUP):
            call(i)
        times: list[int] = []
        engine_us: list[float] = []
        for i in range(N):
            t0 = time.perf_counter_ns()
            res = call(i)
            times.append(time.perf_counter_ns() - t0)
            engine_us.append(res["latency_us"])
        conn.close()
        summarize("POST /v1/evaluate round trip", times)
        summarize("server-side engine latency_us", [int(u * 1000) for u in engine_us])
    finally:
        stop_server(server, thread)


if __name__ == "__main__":
    main()
