view mrjunejune/conversation_api.c @ 260:1f9877b637e9

Add Copilot-powered cyberpunk JRPG chat Integrate the production JRPG chat with Seobeo streaming, Deita persistence, and a Bazel-managed Copilot SDK and LiteLLM inference stack. Co-authored-by: Copilot <[email protected]>
author MrJuneJune <mrjunejune@users.noreply.github.com>
date Wed, 05 Aug 2026 09:19:41 -0700
parents
children b401627fc49e
line wrap: on
line source

#include "mrjunejune/conversation_api.h"

#include "mrjunejune/inference_bridge.h"
#include "mrjunejune/conversation_store.h"
#include "seobeo/seobeo.h"

#include <pthread.h>
#include <stdio.h>
#include <stdarg.h>
#include <stdlib.h>
#include <string.h>
#include <strings.h>
#include <time.h>

#define CONVERSATION_TITLE_MAX 200
#define CONVERSATION_PROMPT_MAX (32 * 1024)
#define CONVERSATION_RESPONSE_MAX (256 * 1024)
#define CONVERSATION_HISTORY_MAX (512 * 1024)
#define CONVERSATION_ACTIVE_MAX 4
#define CONVERSATION_TURNS_PER_MINUTE 60

static Conversation_Store *g_conversation_store = NULL;
static Inference_Bridge *g_inference_bridge = NULL;
static boolean g_anonymous_inference_enabled = FALSE;
static pthread_mutex_t g_pending_mutex = PTHREAD_MUTEX_INITIALIZER;
static pthread_mutex_t g_admission_mutex = PTHREAD_MUTEX_INITIALIZER;
static time_t g_admission_window = 0;
static uint32 g_admission_turns = 0;
static uint32 g_active_turns = 0;

typedef struct Pending_Turn {
  char request_id[37];
  char conversation_id[37];
  Seobeo_SSE_Stream *p_stream;
  char *content;
  size_t content_length;
  size_t content_capacity;
  int64 input_tokens;
  int64 output_tokens;
  boolean failed;
  boolean aborted;
  char error_message[512];
  struct Pending_Turn *p_next;
} Pending_Turn;

static Pending_Turn *g_pending_turns = NULL;

static boolean Conversation_API_Acquire_Turn_Slot(void)
{
  time_t now = time(NULL);
  pthread_mutex_lock(&g_admission_mutex);
  if (g_admission_window == 0 || now - g_admission_window >= 60)
  {
    g_admission_window = now;
    g_admission_turns = 0;
  }
  boolean allowed =
      g_admission_turns < CONVERSATION_TURNS_PER_MINUTE &&
      g_active_turns < CONVERSATION_ACTIVE_MAX;
  if (allowed)
  {
    g_admission_turns++;
    g_active_turns++;
  }
  pthread_mutex_unlock(&g_admission_mutex);
  return allowed;
}

static void Conversation_API_Release_Turn_Slot(void)
{
  pthread_mutex_lock(&g_admission_mutex);
  if (g_active_turns > 0)
    g_active_turns--;
  pthread_mutex_unlock(&g_admission_mutex);
}

static Seobeo_Request_Entry *Conversation_API_JSON_Response(
    Dowa_Arena *p_arena,
    const char *status,
    const char *body)
{
  Seobeo_Request_Entry *p_response = NULL;
  Dowa_HashMap_Push_Arena(p_response, "status", (char *)status, p_arena);
  Dowa_HashMap_Push_Arena(
      p_response, "content-type", "application/json; charset=utf-8", p_arena);
  Dowa_HashMap_Push_Arena(p_response, "cache-control", "no-store", p_arena);
  Dowa_HashMap_Push_Arena(p_response, "body", (char *)body, p_arena);
  return p_response;
}

static Seobeo_Request_Entry *Conversation_API_Error(
    Dowa_Arena *p_arena,
    const char *status,
    const char *code,
    const char *message)
{
  char *escaped = Dowa_JSON_Escape_String(message, 0, p_arena);
  size_t capacity = strlen(code) + strlen(escaped) + 64;
  char *body = Dowa_Arena_Allocate(p_arena, capacity);
  snprintf(
      body,
      capacity,
      "{\"error\":{\"code\":\"%s\",\"message\":\"%s\"}}",
      code,
      escaped);
  return Conversation_API_JSON_Response(p_arena, status, body);
}

static const char *Conversation_API_Request_Value(
    Seobeo_Request_Entry *p_request,
    const char *key)
{
  void *p_value = Dowa_HashMap_Get_Ptr(p_request, (char *)key);
  return p_value ? ((Seobeo_Request_Entry *)p_value)->value : NULL;
}

static boolean Conversation_API_Is_Same_Origin(
    Seobeo_Request_Entry *p_request)
{
  const char *host = Conversation_API_Request_Value(p_request, "Host");
  const char *origin = Conversation_API_Request_Value(p_request, "Origin");
  if (!host || !origin)
    return FALSE;
  const char *origin_host = strstr(origin, "://");
  if (!origin_host)
    return FALSE;
  origin_host += 3;
  const char *end = strchr(origin_host, '/');
  size_t length = end ? (size_t)(end - origin_host) : strlen(origin_host);
  return strlen(host) == length && strncmp(host, origin_host, length) == 0;
}

static boolean Conversation_API_Is_Enabled(void)
{
  return g_anonymous_inference_enabled;
}

static Dowa_JSON_Entry *Conversation_API_Parse_Body(
    Seobeo_Request_Entry *p_request,
    Dowa_Arena *p_arena)
{
  const char *body = Conversation_API_Request_Value(p_request, "Body");
  if (!body)
    return NULL;
  Dowa_JSON_Value value = Dowa_JSON_Parse(
      body, (int32)strlen(body), p_arena);
  return value.type == DOWA_JSON_OBJECT
      ? (Dowa_JSON_Entry *)value.object_val
      : NULL;
}

static void Conversation_API_Send_Stream_Error(
    Seobeo_Handle *p_handle,
    int status,
    const char *code,
    const char *message)
{
  char body[1024];
  int body_length = snprintf(
      body,
      sizeof(body),
      "{\"error\":{\"code\":\"%s\",\"message\":\"%s\"}}",
      code,
      message);
  if (body_length < 0 || (size_t)body_length >= sizeof(body))
    return;
  char header[512];
  Seobeo_Web_Header_Generate(
      header,
      status,
      "application/json; charset=utf-8",
      body_length);
  Seobeo_Handle_Queue(
      p_handle, (const uint8 *)header, (uint32)strlen(header));
  Seobeo_Handle_Queue(
      p_handle, (const uint8 *)body, (uint32)body_length);
  Seobeo_Handle_Flush(p_handle);
}

static Pending_Turn *Conversation_API_Find_Pending(
    const char *request_id)
{
  for (Pending_Turn *p_turn = g_pending_turns;
       p_turn;
       p_turn = p_turn->p_next)
  {
    if (strcmp(p_turn->request_id, request_id) == 0)
      return p_turn;
  }
  return NULL;
}

static int32 Conversation_API_Send_Event(
    Pending_Turn *p_turn,
    const char *event_name,
    const char *json)
{
  Seobeo_SSE_Event event = {
    .event = event_name,
    .data = json,
    .retry_ms = SEOBEO_SSE_RETRY_NONE,
  };
  return Seobeo_SSE_Send(p_turn->p_stream, &event);
}

static void Conversation_API_Remove_Pending(Pending_Turn *p_turn)
{
  Pending_Turn **pp_current = &g_pending_turns;
  while (*pp_current)
  {
    if (*pp_current == p_turn)
    {
      *pp_current = p_turn->p_next;
      return;
    }
    pp_current = &(*pp_current)->p_next;
  }
}

static void Conversation_API_Finalize_Pending(Pending_Turn *p_turn)
{
  Conversation_Store_Result persistence;
  if (p_turn->failed || p_turn->aborted)
  {
    persistence = Conversation_Store_Fail_Turn(
        g_conversation_store,
        p_turn->conversation_id,
        p_turn->request_id,
        p_turn->error_message,
        p_turn->aborted);
  }
  else
  {
    persistence = Conversation_Store_Complete_Turn(
        g_conversation_store,
        p_turn->conversation_id,
        p_turn->request_id,
        p_turn->content ? p_turn->content : "",
        p_turn->input_tokens,
        p_turn->output_tokens);
  }
  if (persistence != CONVERSATION_STORE_OK)
  {
    p_turn->failed = TRUE;
    snprintf(
        p_turn->error_message,
        sizeof(p_turn->error_message),
        "Unable to persist completed turn");
    Conversation_Store_Fail_Turn(
        g_conversation_store,
        p_turn->conversation_id,
        p_turn->request_id,
        p_turn->error_message,
        FALSE);
    Conversation_API_Send_Event(
        p_turn,
        "turn.error",
        "{\"code\":\"storage_failed\","
        "\"message\":\"Unable to persist completed turn\"}");
  }

  char done[256];
  snprintf(
      done,
      sizeof(done),
      "{\"request_id\":\"%s\",\"failed\":%s,\"aborted\":%s}",
      p_turn->request_id,
      p_turn->failed ? "true" : "false",
      p_turn->aborted ? "true" : "false");
  Conversation_API_Send_Event(p_turn, "turn.done", done);
  Conversation_API_Remove_Pending(p_turn);
  Seobeo_SSE_Close(p_turn->p_stream);
  Seobeo_SSE_Release(p_turn->p_stream);
  Conversation_API_Release_Turn_Slot();
  free(p_turn->content);
  free(p_turn);
}

static void Conversation_API_Handle_Inference_Event(
    const Inference_Event *p_event,
    void *p_user_data)
{
  (void)p_user_data;
  if (strcmp(p_event->type, "bridge.closed") == 0)
  {
    pthread_mutex_lock(&g_pending_mutex);
    while (g_pending_turns)
    {
      Pending_Turn *p_turn = g_pending_turns;
      p_turn->failed = TRUE;
      snprintf(
          p_turn->error_message,
          sizeof(p_turn->error_message),
          "Inference sidecar connection closed");
      Conversation_API_Send_Event(
          p_turn,
          "turn.error",
          "{\"code\":\"sidecar_closed\","
          "\"message\":\"Inference sidecar connection closed\"}");
      Conversation_API_Finalize_Pending(p_turn);
    }
    pthread_mutex_unlock(&g_pending_mutex);
    return;
  }
  if (!p_event->request_id || p_event->request_id[0] == '\0')
    return;

  boolean should_abort = FALSE;
  char abort_request_id[37] = {0};
  char abort_conversation_id[37] = {0};
  pthread_mutex_lock(&g_pending_mutex);
  Pending_Turn *p_turn = Conversation_API_Find_Pending(p_event->request_id);
  if (!p_turn)
  {
    pthread_mutex_unlock(&g_pending_mutex);
    return;
  }

  size_t event_text_length =
      strlen(p_event->delta ? p_event->delta : "") +
      strlen(p_event->content ? p_event->content : "") +
      strlen(p_event->error_code ? p_event->error_code : "") +
      strlen(p_event->error_message ? p_event->error_message : "");
  if (event_text_length > (((size_t)-1) - 8192) / 12)
  {
    p_turn->failed = TRUE;
    snprintf(
        p_turn->error_message,
        sizeof(p_turn->error_message),
        "Inference event exceeded memory limit");
    Conversation_API_Send_Event(
        p_turn,
        "turn.error",
        "{\"code\":\"event_too_large\","
        "\"message\":\"Inference event exceeded memory limit\"}");
    Conversation_API_Finalize_Pending(p_turn);
    pthread_mutex_unlock(&g_pending_mutex);
    return;
  }
  Dowa_Arena *p_arena = Dowa_Arena_Create(
      event_text_length * 12 + 8192);
  if (!p_arena)
  {
    p_turn->failed = TRUE;
    snprintf(
        p_turn->error_message,
        sizeof(p_turn->error_message),
        "Unable to allocate inference event");
    Conversation_API_Send_Event(
        p_turn,
        "turn.error",
        "{\"code\":\"allocation_failed\","
        "\"message\":\"Unable to allocate inference event\"}");
    Conversation_API_Finalize_Pending(p_turn);
    pthread_mutex_unlock(&g_pending_mutex);
    return;
  }

  if (strcmp(p_event->type, "turn.accepted") == 0)
  {
    char accepted[128];
    snprintf(
        accepted,
        sizeof(accepted),
        "{\"request_id\":\"%s\"}",
        p_turn->request_id);
    Conversation_API_Send_Event(p_turn, "turn.accepted", accepted);
  }
  else if (strcmp(p_event->type, "assistant.delta") == 0)
  {
    size_t delta_length = strlen(p_event->delta);
    if (p_turn->content_length + delta_length > CONVERSATION_RESPONSE_MAX)
    {
      p_turn->failed = TRUE;
      snprintf(
          p_turn->error_message,
          sizeof(p_turn->error_message),
          "Assistant response exceeded limit");
      should_abort = TRUE;
      snprintf(abort_request_id, sizeof(abort_request_id), "%s",
               p_turn->request_id);
      snprintf(abort_conversation_id, sizeof(abort_conversation_id), "%s",
               p_turn->conversation_id);
      char error[256];
      snprintf(
          error,
          sizeof(error),
          "{\"request_id\":\"%s\",\"code\":\"response_too_large\","
          "\"message\":\"Assistant response exceeded limit\"}",
          p_turn->request_id);
      Conversation_API_Send_Event(p_turn, "turn.error", error);
    }
    else
    {
      if (p_turn->content_length + delta_length + 1 >
          p_turn->content_capacity)
      {
        size_t next_capacity = p_turn->content_capacity
            ? p_turn->content_capacity * 2
            : 4096;
        while (next_capacity <
               p_turn->content_length + delta_length + 1)
          next_capacity *= 2;
        char *next = realloc(p_turn->content, next_capacity);
        if (!next)
        {
          p_turn->failed = TRUE;
          snprintf(
              p_turn->error_message,
              sizeof(p_turn->error_message),
              "Unable to buffer assistant response");
        }
        else
        {
          p_turn->content = next;
          p_turn->content_capacity = next_capacity;
        }
      }
      if (!p_turn->failed)
      {
        memcpy(
            p_turn->content + p_turn->content_length,
            p_event->delta,
            delta_length);
        p_turn->content_length += delta_length;
        p_turn->content[p_turn->content_length] = '\0';
        char *delta = Dowa_JSON_Escape_String(
            p_event->delta, delta_length, p_arena);
        size_t capacity = delta ? strlen(delta) + 128 : 0;
        char *json = capacity
            ? Dowa_Arena_Allocate(p_arena, capacity)
            : NULL;
        if (!delta || !json)
        {
          p_turn->failed = TRUE;
          snprintf(
              p_turn->error_message,
              sizeof(p_turn->error_message),
              "Unable to serialize assistant delta");
          Conversation_API_Send_Event(
              p_turn,
              "turn.error",
              "{\"code\":\"serialize_failed\","
              "\"message\":\"Unable to serialize assistant delta\"}");
          should_abort = TRUE;
          snprintf(abort_request_id, sizeof(abort_request_id), "%s",
                   p_turn->request_id);
          snprintf(abort_conversation_id, sizeof(abort_conversation_id), "%s",
                   p_turn->conversation_id);
        }
        else
        {
          snprintf(
              json,
              capacity,
              "{\"request_id\":\"%s\",\"delta\":\"%s\"}",
              p_turn->request_id,
              delta);
          if (Conversation_API_Send_Event(
                  p_turn, "assistant.delta", json) < 0)
          {
            p_turn->aborted = TRUE;
            should_abort = TRUE;
            snprintf(abort_request_id, sizeof(abort_request_id), "%s",
                     p_turn->request_id);
            snprintf(abort_conversation_id, sizeof(abort_conversation_id), "%s",
                     p_turn->conversation_id);
          }
        }
      }
    }
  }
  else if (strcmp(p_event->type, "assistant.completed") == 0)
  {
    size_t content_length = strlen(p_event->content);
    if (content_length > CONVERSATION_RESPONSE_MAX)
    {
      p_turn->failed = TRUE;
      snprintf(
          p_turn->error_message,
          sizeof(p_turn->error_message),
          "Assistant response exceeded limit");
      char error[256];
      snprintf(
          error,
          sizeof(error),
          "{\"request_id\":\"%s\",\"code\":\"response_too_large\","
          "\"message\":\"Assistant response exceeded limit\"}",
          p_turn->request_id);
      Conversation_API_Send_Event(p_turn, "turn.error", error);
    }
    else
    {
      char *content = realloc(p_turn->content, content_length + 1);
      if (content)
      {
        p_turn->content = content;
        p_turn->content_capacity = content_length + 1;
        memcpy(p_turn->content, p_event->content, content_length + 1);
        p_turn->content_length = content_length;
      }
      else
      {
        p_turn->failed = TRUE;
        snprintf(
            p_turn->error_message,
            sizeof(p_turn->error_message),
            "Unable to buffer assistant response");
      }
      if (!p_turn->failed)
      {
        char *escaped = Dowa_JSON_Escape_String(
            p_event->content, content_length, p_arena);
        size_t capacity = escaped ? strlen(escaped) + 128 : 0;
        char *json = capacity
            ? Dowa_Arena_Allocate(p_arena, capacity)
            : NULL;
        if (!escaped || !json)
        {
          p_turn->failed = TRUE;
          snprintf(
              p_turn->error_message,
              sizeof(p_turn->error_message),
              "Unable to serialize assistant response");
          Conversation_API_Send_Event(
              p_turn,
              "turn.error",
              "{\"code\":\"serialize_failed\","
              "\"message\":\"Unable to serialize assistant response\"}");
        }
        else
        {
          snprintf(
              json,
              capacity,
              "{\"request_id\":\"%s\",\"content\":\"%s\"}",
              p_turn->request_id,
              escaped);
          Conversation_API_Send_Event(p_turn, "assistant.completed", json);
        }
      }
    }
  }
  else if (strcmp(p_event->type, "assistant.usage") == 0)
  {
    p_turn->input_tokens = p_event->input_tokens;
    p_turn->output_tokens = p_event->output_tokens;
    char usage[256];
    snprintf(
        usage,
        sizeof(usage),
        "{\"request_id\":\"%s\",\"input_tokens\":%lld,"
        "\"output_tokens\":%lld}",
        p_turn->request_id,
        (long long)p_turn->input_tokens,
        (long long)p_turn->output_tokens);
    Conversation_API_Send_Event(p_turn, "assistant.usage", usage);
  }
  else if (strcmp(p_event->type, "turn.error") == 0)
  {
    p_turn->failed = TRUE;
    snprintf(
        p_turn->error_message,
        sizeof(p_turn->error_message),
        "%s",
        p_event->error_message);
    char *code = Dowa_JSON_Escape_String(
        p_event->error_code, 0, p_arena);
    char *message = Dowa_JSON_Escape_String(
        p_event->error_message, 0, p_arena);
    size_t capacity = code && message
        ? strlen(code) + strlen(message) + 160
        : 0;
    char *json = capacity
        ? Dowa_Arena_Allocate(p_arena, capacity)
        : NULL;
    if (!code || !message || !json)
    {
      Conversation_API_Send_Event(
          p_turn,
          "turn.error",
          "{\"code\":\"serialize_failed\","
          "\"message\":\"Unable to serialize inference error\"}");
    }
    else
    {
      snprintf(
          json,
          capacity,
          "{\"request_id\":\"%s\",\"code\":\"%s\",\"message\":\"%s\"}",
          p_turn->request_id,
          code,
          message);
      Conversation_API_Send_Event(p_turn, "turn.error", json);
    }
  }
  else if (strcmp(p_event->type, "turn.done") == 0)
  {
    p_turn->failed = p_turn->failed || p_event->failed;
    p_turn->aborted = p_turn->aborted || p_event->aborted;
    Conversation_API_Finalize_Pending(p_turn);
  }

  Dowa_Arena_Free(p_arena);
  pthread_mutex_unlock(&g_pending_mutex);
  if (should_abort)
    Inference_Bridge_Abort_Turn(
        g_inference_bridge, abort_request_id, abort_conversation_id);
}

static Seobeo_Request_Entry *Conversation_API_Create(
    Seobeo_Request_Entry *p_request,
    Dowa_Arena *p_arena)
{
  if (!Conversation_API_Is_Enabled())
    return Conversation_API_Error(
        p_arena, "503", "inference_disabled", "Inference API is disabled");
  if (!Conversation_API_Is_Same_Origin(p_request))
    return Conversation_API_Error(
        p_arena, "403", "origin_rejected", "Same-origin request required");
  if (!g_conversation_store)
    return Conversation_API_Error(
        p_arena, "503", "store_unavailable", "Conversation store unavailable");

  const char *title = "";
  const char *body = Conversation_API_Request_Value(p_request, "Body");
  if (body && body[0] != '\0')
  {
    Dowa_JSON_Entry *object = Conversation_API_Parse_Body(p_request, p_arena);
    if (!object)
      return Conversation_API_Error(
          p_arena, "400", "invalid_json", "Request body must be a JSON object");
    char *parsed_title = Dowa_JSON_Get_String(object, "title");
    if (parsed_title)
      title = parsed_title;
  }
  if (strlen(title) > CONVERSATION_TITLE_MAX)
    return Conversation_API_Error(
        p_arena, "400", "invalid_title", "Title exceeds 200 bytes");

  char conversation_id[37];
  if (Conversation_Store_Create_Conversation(
          g_conversation_store,
          title,
          conversation_id) != CONVERSATION_STORE_OK)
    return Conversation_API_Error(
        p_arena, "500", "create_failed", "Unable to create conversation");

  char *response_body = Dowa_Arena_Allocate(p_arena, 64);
  snprintf(response_body, 64, "{\"id\":\"%s\"}", conversation_id);
  return Conversation_API_JSON_Response(p_arena, "201", response_body);
}

static boolean Conversation_API_Append(
    char *output,
    size_t capacity,
    size_t *p_offset,
    const char *format,
    ...)
{
  if (*p_offset >= capacity)
    return FALSE;
  va_list args;
  va_start(args, format);
  int written = vsnprintf(
      output + *p_offset, capacity - *p_offset, format, args);
  va_end(args);
  if (written < 0 || (size_t)written >= capacity - *p_offset)
    return FALSE;
  *p_offset += (size_t)written;
  return TRUE;
}

static Seobeo_Request_Entry *Conversation_API_Get(
    Seobeo_Request_Entry *p_request,
    Dowa_Arena *p_arena)
{
  if (!Conversation_API_Is_Enabled())
    return Conversation_API_Error(
        p_arena, "503", "inference_disabled", "Inference API is disabled");
  const char *conversation_id =
      Conversation_API_Request_Value(p_request, ":conversation_id");
  if (!conversation_id)
    return Conversation_API_Error(
        p_arena, "400", "missing_id", "Conversation ID is required");

  Conversation_Record record;
  Conversation_Store_Result result = Conversation_Store_Get(
      g_conversation_store, conversation_id, &record, p_arena);
  if (result == CONVERSATION_STORE_NOT_FOUND)
    return Conversation_API_Error(
        p_arena, "404", "not_found", "Conversation not found");
  if (result != CONVERSATION_STORE_OK)
    return Conversation_API_Error(
        p_arena, "500", "load_failed", "Unable to load conversation");

  if (!record.id || !record.title || !record.status)
    return Conversation_API_Error(
        p_arena, "500", "load_failed", "Conversation data exceeded limits");
  size_t history_size = strlen(record.title) + strlen(record.status);
  for (size_t i = 0; i < Dowa_Array_Length(record.turns); i++)
  {
    Conversation_Turn *p_turn = &record.turns[i];
    if (!p_turn->role || !p_turn->content || !p_turn->status ||
        !p_turn->request_id || !p_turn->error_message)
      return Conversation_API_Error(
          p_arena, "500", "load_failed", "Conversation data exceeded limits");
    history_size +=
        strlen(p_turn->role) + strlen(p_turn->content) +
        strlen(p_turn->status) + strlen(p_turn->request_id) +
        strlen(p_turn->error_message);
    if (history_size > CONVERSATION_HISTORY_MAX)
      return Conversation_API_Error(
          p_arena,
          "413",
          "history_too_large",
          "Conversation history exceeds response limit");
  }
  char *escaped_title = Dowa_JSON_Escape_String(record.title, 0, p_arena);
  if (!escaped_title)
    return Conversation_API_Error(
        p_arena, "500", "serialize_failed", "Unable to serialize conversation");
  size_t capacity = strlen(escaped_title) + 512;
  for (size_t i = 0; i < Dowa_Array_Length(record.turns); i++)
  {
    Conversation_Turn *p_turn = &record.turns[i];
    capacity +=
        (strlen(p_turn->role) + strlen(p_turn->content) +
         strlen(p_turn->status) + strlen(p_turn->request_id) +
         strlen(p_turn->error_message)) * 6 + 320;
  }
  char *body = Dowa_Arena_Allocate(p_arena, capacity);
  if (!body)
    return Conversation_API_Error(
        p_arena, "500", "serialize_failed", "Response exceeds memory limit");
  size_t offset = 0;
  if (!Conversation_API_Append(
          body,
          capacity,
          &offset,
          "{\"id\":\"%s\",\"title\":\"%s\",\"status\":\"%s\","
          "\"created_at\":%lld,\"updated_at\":%lld,\"turns\":[",
          record.id,
          escaped_title,
          record.status,
          (long long)record.created_at,
          (long long)record.updated_at))
    return Conversation_API_Error(
        p_arena, "500", "serialize_failed", "Response too large");

  for (size_t i = 0; i < Dowa_Array_Length(record.turns); i++)
  {
    Conversation_Turn *p_turn = &record.turns[i];
    char *role = Dowa_JSON_Escape_String(p_turn->role, 0, p_arena);
    char *content = Dowa_JSON_Escape_String(p_turn->content, 0, p_arena);
    char *status = Dowa_JSON_Escape_String(p_turn->status, 0, p_arena);
    char *request_id = Dowa_JSON_Escape_String(
        p_turn->request_id, 0, p_arena);
    char *error = Dowa_JSON_Escape_String(
        p_turn->error_message, 0, p_arena);
    if (!Conversation_API_Append(
            body,
            capacity,
            &offset,
            "%s{\"id\":%lld,\"sequence\":%lld,\"role\":\"%s\","
            "\"content\":\"%s\",\"status\":\"%s\","
            "\"request_id\":\"%s\",\"error\":\"%s\","
            "\"input_tokens\":%lld,\"output_tokens\":%lld,"
            "\"created_at\":%lld,\"completed_at\":%lld}",
            i == 0 ? "" : ",",
            (long long)p_turn->id,
            (long long)p_turn->sequence,
            role,
            content,
            status,
            request_id,
            error,
            (long long)p_turn->input_tokens,
            (long long)p_turn->output_tokens,
            (long long)p_turn->created_at,
            (long long)p_turn->completed_at))
      return Conversation_API_Error(
          p_arena, "500", "serialize_failed", "Response too large");
  }
  if (!Conversation_API_Append(body, capacity, &offset, "]}"))
    return Conversation_API_Error(
        p_arena, "500", "serialize_failed", "Response too large");
  return Conversation_API_JSON_Response(p_arena, "200", body);
}

static Seobeo_Request_Entry *Conversation_API_Update(
    Seobeo_Request_Entry *p_request,
    Dowa_Arena *p_arena)
{
  if (!Conversation_API_Is_Enabled())
    return Conversation_API_Error(
        p_arena, "503", "inference_disabled", "Inference API is disabled");
  if (!Conversation_API_Is_Same_Origin(p_request))
    return Conversation_API_Error(
        p_arena, "403", "origin_rejected", "Same-origin request required");
  const char *conversation_id =
      Conversation_API_Request_Value(p_request, ":conversation_id");
  Dowa_JSON_Entry *object = Conversation_API_Parse_Body(p_request, p_arena);
  char *title = object ? Dowa_JSON_Get_String(object, "title") : NULL;
  if (!conversation_id || !title)
    return Conversation_API_Error(
        p_arena, "400", "invalid_request", "Conversation ID and title required");
  if (strlen(title) > CONVERSATION_TITLE_MAX)
    return Conversation_API_Error(
        p_arena, "400", "invalid_title", "Title exceeds 200 bytes");

  Conversation_Store_Result result = Conversation_Store_Update_Title(
      g_conversation_store, conversation_id, title);
  if (result == CONVERSATION_STORE_NOT_FOUND)
    return Conversation_API_Error(
        p_arena, "404", "not_found", "Conversation not found");
  if (result != CONVERSATION_STORE_OK)
    return Conversation_API_Error(
        p_arena, "500", "update_failed", "Unable to update conversation");
  return Conversation_API_JSON_Response(p_arena, "200", "{\"ok\":true}");
}

static Seobeo_Request_Entry *Conversation_API_Delete(
    Seobeo_Request_Entry *p_request,
    Dowa_Arena *p_arena)
{
  if (!Conversation_API_Is_Enabled())
    return Conversation_API_Error(
        p_arena, "503", "inference_disabled", "Inference API is disabled");
  if (!Conversation_API_Is_Same_Origin(p_request))
    return Conversation_API_Error(
        p_arena, "403", "origin_rejected", "Same-origin request required");
  const char *conversation_id =
      Conversation_API_Request_Value(p_request, ":conversation_id");
  if (!conversation_id)
    return Conversation_API_Error(
        p_arena, "400", "missing_id", "Conversation ID is required");
  Conversation_Store_Result result = Conversation_Store_Delete(
      g_conversation_store, conversation_id);
  if (result == CONVERSATION_STORE_NOT_FOUND)
    return Conversation_API_Error(
        p_arena, "404", "not_found", "Conversation not found");
  if (result != CONVERSATION_STORE_OK)
    return Conversation_API_Error(
        p_arena, "500", "delete_failed", "Unable to delete conversation");
  if (Inference_Bridge_Is_Ready(g_inference_bridge))
  {
    char request_id[37];
    if (Conversation_Store_Generate_UUID(request_id))
      Inference_Bridge_Delete_Conversation(
          g_inference_bridge, request_id, conversation_id);
  }

  Seobeo_Request_Entry *p_response = NULL;
  Dowa_HashMap_Push_Arena(p_response, "status", "204", p_arena);
  Dowa_HashMap_Push_Arena(p_response, "body", "", p_arena);
  return p_response;
}

static Seobeo_Request_Entry *Conversation_API_Health(
    Seobeo_Request_Entry *p_request,
    Dowa_Arena *p_arena)
{
  (void)p_request;
  boolean ready =
      Conversation_API_Is_Enabled() &&
      g_conversation_store &&
      Inference_Bridge_Is_Ready(g_inference_bridge);
  return Conversation_API_JSON_Response(
      p_arena,
      ready ? "200" : "503",
      ready ? "{\"status\":\"ready\"}" : "{\"status\":\"unavailable\"}");
}

static void Conversation_API_Turn_Stream(
    Seobeo_Handle *p_handle,
    Seobeo_Request_Entry *p_request,
    Dowa_Arena *p_arena)
{
  if (!Conversation_API_Is_Enabled())
  {
    Conversation_API_Send_Stream_Error(
        p_handle, 503, "inference_disabled", "Inference API is disabled");
    return;
  }
  if (!Conversation_API_Is_Same_Origin(p_request))
  {
    Conversation_API_Send_Stream_Error(
        p_handle, 403, "origin_rejected", "Same-origin request required");
    return;
  }
  if (!g_conversation_store || !Inference_Bridge_Is_Ready(g_inference_bridge))
  {
    Conversation_API_Send_Stream_Error(
        p_handle, 503, "inference_unavailable", "Inference runtime unavailable");
    return;
  }
  const char *conversation_id =
      Conversation_API_Request_Value(p_request, ":conversation_id");
  Dowa_JSON_Entry *object = Conversation_API_Parse_Body(p_request, p_arena);
  char *prompt = object ? Dowa_JSON_Get_String(object, "prompt") : NULL;
  if (!conversation_id || !prompt || prompt[0] == '\0')
  {
    Conversation_API_Send_Stream_Error(
        p_handle, 400, "invalid_request", "Conversation ID and prompt required");
    return;
  }
  size_t prompt_length = strlen(prompt);
  if (prompt_length > CONVERSATION_PROMPT_MAX)
  {
    Conversation_API_Send_Stream_Error(
        p_handle, 413, "prompt_too_large", "Prompt exceeds 32 KiB");
    return;
  }
  if (!Conversation_API_Acquire_Turn_Slot())
  {
    Conversation_API_Send_Stream_Error(
        p_handle, 429, "rate_limited", "Inference capacity exhausted");
    return;
  }

  char request_id[37];
  if (!Conversation_Store_Generate_UUID(request_id))
  {
    Conversation_API_Release_Turn_Slot();
    Conversation_API_Send_Stream_Error(
        p_handle, 500, "id_failed", "Unable to create request ID");
    return;
  }
  Conversation_Store_Result result = Conversation_Store_Begin_Turn(
      g_conversation_store,
      conversation_id,
      request_id,
      prompt);
  if (result == CONVERSATION_STORE_NOT_FOUND)
  {
    Conversation_API_Release_Turn_Slot();
    Conversation_API_Send_Stream_Error(
        p_handle, 404, "not_found", "Conversation not found");
    return;
  }
  if (result == CONVERSATION_STORE_CONFLICT)
  {
    Conversation_API_Release_Turn_Slot();
    Conversation_API_Send_Stream_Error(
        p_handle, 409, "turn_in_progress", "Conversation already has an active turn");
    return;
  }
  if (result != CONVERSATION_STORE_OK)
  {
    Conversation_API_Release_Turn_Slot();
    Conversation_API_Send_Stream_Error(
        p_handle, 500, "turn_failed", "Unable to persist turn");
    return;
  }

  Seobeo_SSE_Stream *p_stream = Seobeo_SSE_Server_Attach(
      p_handle,
      "/api/conversations/turns");
  if (!p_stream || !Seobeo_SSE_Retain(p_stream))
  {
    Conversation_API_Release_Turn_Slot();
    Conversation_Store_Fail_Turn(
        g_conversation_store,
        conversation_id,
        request_id,
        "Unable to start event stream",
        FALSE);
    if (!p_stream)
      Conversation_API_Send_Stream_Error(
          p_handle, 500, "stream_failed", "Unable to start event stream");
    return;
  }

  Pending_Turn *p_turn = calloc(1, sizeof(*p_turn));
  if (!p_turn)
  {
    Conversation_API_Release_Turn_Slot();
    Conversation_Store_Fail_Turn(
        g_conversation_store,
        conversation_id,
        request_id,
        "Unable to allocate turn",
        FALSE);
    Seobeo_SSE_Send_Data(
        p_stream,
        "{\"error\":{\"code\":\"allocation_failed\"}}");
    Seobeo_SSE_Close(p_stream);
    Seobeo_SSE_Release(p_stream);
    return;
  }
  snprintf(p_turn->request_id, sizeof(p_turn->request_id), "%s", request_id);
  snprintf(
      p_turn->conversation_id,
      sizeof(p_turn->conversation_id),
      "%s",
      conversation_id);
  p_turn->p_stream = p_stream;

  pthread_mutex_lock(&g_pending_mutex);
  p_turn->p_next = g_pending_turns;
  g_pending_turns = p_turn;
  pthread_mutex_unlock(&g_pending_mutex);

  if (!Inference_Bridge_Start_Turn(
          g_inference_bridge, request_id, conversation_id, prompt))
  {
    pthread_mutex_lock(&g_pending_mutex);
    Pending_Turn *p_pending = Conversation_API_Find_Pending(request_id);
    if (p_pending)
    {
      p_pending->failed = TRUE;
      snprintf(
          p_pending->error_message,
          sizeof(p_pending->error_message),
          "Unable to dispatch inference turn");
      Conversation_API_Send_Event(
          p_pending,
          "turn.error",
          "{\"code\":\"dispatch_failed\","
          "\"message\":\"Unable to dispatch inference turn\"}");
      Conversation_API_Finalize_Pending(p_pending);
    }
    pthread_mutex_unlock(&g_pending_mutex);
  }
}

boolean Conversation_API_Init(const char *database_path)
{
  if (g_conversation_store)
    return TRUE;
  const char *allow_anonymous = getenv(
      "MRJUNEJUNE_ALLOW_ANONYMOUS_INFERENCE");
  g_anonymous_inference_enabled =
      allow_anonymous &&
      (strcmp(allow_anonymous, "1") == 0 ||
       strcasecmp(allow_anonymous, "true") == 0);
  g_conversation_store = Conversation_Store_Create(database_path);
  return g_conversation_store != NULL;
}

boolean Conversation_API_Enable_Inference(
    const char *sidecar_path,
    const char *copilot_cli_path)
{
  if (g_inference_bridge)
    return Inference_Bridge_Is_Ready(g_inference_bridge);
  if (!sidecar_path || !copilot_cli_path)
    return FALSE;
  g_inference_bridge = Inference_Bridge_Create(
      sidecar_path,
      copilot_cli_path,
      Conversation_API_Handle_Inference_Event,
      NULL);
  if (!g_inference_bridge)
    return FALSE;
  if (!Inference_Bridge_Start(g_inference_bridge))
  {
    Inference_Bridge_Destroy(g_inference_bridge);
    g_inference_bridge = NULL;
    return FALSE;
  }
  return TRUE;
}

void Conversation_API_Register_Routes(void)
{
  Seobeo_Router_Register(
      "GET", "/api/inference/health", Conversation_API_Health);
  Seobeo_Router_Register("POST", "/api/conversations", Conversation_API_Create);
  Seobeo_Router_Register(
      "GET", "/api/conversations/:conversation_id", Conversation_API_Get);
  Seobeo_Router_Register(
      "PATCH", "/api/conversations/:conversation_id", Conversation_API_Update);
  Seobeo_Router_Register(
      "DELETE", "/api/conversations/:conversation_id", Conversation_API_Delete);
  Seobeo_Router_Register_Stream(
      "POST",
      "/api/conversations/:conversation_id/turns",
      Conversation_API_Turn_Stream);
}

void Conversation_API_Destroy(void)
{
  Inference_Bridge_Destroy(g_inference_bridge);
  g_inference_bridge = NULL;
  pthread_mutex_lock(&g_pending_mutex);
  while (g_pending_turns)
  {
    Pending_Turn *p_turn = g_pending_turns;
    g_pending_turns = p_turn->p_next;
    Conversation_Store_Fail_Turn(
        g_conversation_store,
        p_turn->conversation_id,
        p_turn->request_id,
        "Server shutdown",
        TRUE);
    Seobeo_SSE_Close(p_turn->p_stream);
    Seobeo_SSE_Release(p_turn->p_stream);
    free(p_turn->content);
    free(p_turn);
    Conversation_API_Release_Turn_Slot();
  }
  pthread_mutex_unlock(&g_pending_mutex);
  Conversation_Store_Destroy(g_conversation_store);
  g_conversation_store = NULL;
  g_anonymous_inference_enabled = FALSE;
}