{
  "schemaVersion": "1.0.0",
  "runId": "d141d9ec-12b9-4b4e-b807-3eea5c265ca8",
  "generatedAt": "2026-08-08T18:48:23.057Z",
  "mode": "live",
  "title": "Token Factory Agent Routing Lab",
  "question": "Why did this agent request become slow after I added shared context?",
  "context": "The agent has a large static policy prompt and returns a brief answer. Compare the next run with a stable prompt prefix, then separate observed API evidence from platform-level recommendations.",
  "disclosure": "This is one controlled observation on shared endpoints. It is not a latency, quality, cache, or reliability benchmark.",
  "mcp": {
    "transport": "stdio",
    "server": {
      "name": "routing-lab-context",
      "version": "1.0.0"
    },
    "connectMs": 187,
    "tools": [
      "get_controlled_context"
    ]
  },
  "routes": [
    {
      "id": "A",
      "label": "Economy route",
      "plannerModel": {
        "id": "Qwen/Qwen3-30B-A3B-Instruct-2507",
        "name": "Qwen3-30B-A3B-Instruct-2507",
        "promptPricePerMillion": 0.1,
        "completionPricePerMillion": 0.3,
        "features": [
          "tools",
          "json_mode",
          "structured_outputs"
        ],
        "region": "eu-north1"
      },
      "answerModel": {
        "id": "Qwen/Qwen3-30B-A3B-Instruct-2507",
        "name": "Qwen3-30B-A3B-Instruct-2507",
        "promptPricePerMillion": 0.1,
        "completionPricePerMillion": 0.3,
        "features": [
          "tools",
          "json_mode",
          "structured_outputs"
        ],
        "region": "eu-north1"
      },
      "answer": "**Observed:** The agent’s request slowed after adding shared context, despite a stable prompt prefix and brief output. The large static policy prompt likely increases payload size, causing latency in transmission and processing. Shared context may amplify token count, affecting API response time.\n\n**Recommendation:** Measure token count before and after adding shared context. Identify if the increase correlates with slowdown.\n\n**Next Action:** Use a token counter to compare the total input tokens with and without shared context.",
      "observedTotalMs": 1935,
      "observedFirstTokenMs": 283,
      "estimatedCostUsd": 0.0000983,
      "stages": [
        {
          "name": "plan",
          "model": "Qwen/Qwen3-30B-A3B-Instruct-2507",
          "elapsedMs": 689,
          "firstTokenMs": null,
          "usage": {
            "inputTokens": 339,
            "outputTokens": 39,
            "totalTokens": 378,
            "cachedTokens": 336
          },
          "estimatedCostUsd": 0.0000456,
          "attempts": [
            {
              "attempt": 1,
              "status": 200,
              "elapsedMs": 689,
              "outcome": "success"
            }
          ],
          "status": "passed",
          "detail": "Planner emitted the MCP tool request."
        },
        {
          "name": "mcp_tool",
          "tool": "get_controlled_context",
          "elapsedMs": 2,
          "firstTokenMs": null,
          "usage": null,
          "estimatedCostUsd": 0,
          "attempts": [
            {
              "attempt": 1,
              "status": 200,
              "elapsedMs": 2,
              "outcome": "success"
            }
          ],
          "status": "passed",
          "detail": "This MCP tool returns only the text supplied in the current run. It does not browse, infer telemetry, or access local files."
        },
        {
          "name": "answer",
          "model": "Qwen/Qwen3-30B-A3B-Instruct-2507",
          "elapsedMs": 1244,
          "firstTokenMs": 283,
          "usage": {
            "inputTokens": 242,
            "outputTokens": 95,
            "totalTokens": 337,
            "cachedTokens": 32
          },
          "estimatedCostUsd": 0.0000527,
          "attempts": [
            {
              "attempt": 1,
              "status": 200,
              "elapsedMs": 1244,
              "outcome": "success"
            }
          ],
          "status": "passed",
          "detail": "Streaming completion finished with stop."
        }
      ],
      "reliabilityEvent": null
    },
    {
      "id": "B",
      "label": "Capability route",
      "plannerModel": {
        "id": "Qwen/Qwen3-30B-A3B-Instruct-2507",
        "name": "Qwen3-30B-A3B-Instruct-2507",
        "promptPricePerMillion": 0.1,
        "completionPricePerMillion": 0.3,
        "features": [
          "tools",
          "json_mode",
          "structured_outputs"
        ],
        "region": "eu-north1"
      },
      "answerModel": {
        "id": "openai/gpt-oss-120b",
        "name": "gpt-oss-120b",
        "promptPricePerMillion": 0.15,
        "completionPricePerMillion": 0.6,
        "features": [
          "tools",
          "json_mode",
          "structured_outputs",
          "reasoning"
        ],
        "region": "eu-north1"
      },
      "answer": "**Observed context**  \nThe request now includes a shared‑context block that is concatenated to the static policy prompt. The block contains the full conversation history (or large data payload) each time the agent runs. Because the model must re‑process this extra text on every call, token count and latency increase proportionally, making the request noticeably slower.\n\n**Recommendation**  \nMove the shared context out of the prompt and store it separately (e.g., in a vector store or external cache). Retrieve only the minimal relevant snippets for each call, or use the model’s “system‑message” or “memory” features that don’t count toward token limits.\n\n**Next action**  \nImplement a retrieval step that selects the top‑k relevant pieces from the shared context and appends only those to the prompt before invoking the model. This will cut token usage and restore speed.",
      "observedTotalMs": 1947,
      "observedFirstTokenMs": 234,
      "estimatedCostUsd": 0.00019845,
      "stages": [
        {
          "name": "plan",
          "model": "Qwen/Qwen3-30B-A3B-Instruct-2507",
          "elapsedMs": 1040,
          "firstTokenMs": null,
          "usage": {
            "inputTokens": 339,
            "outputTokens": 39,
            "totalTokens": 378,
            "cachedTokens": 336
          },
          "estimatedCostUsd": 0.0000456,
          "attempts": [
            {
              "attempt": 1,
              "status": 200,
              "elapsedMs": 1040,
              "outcome": "success"
            }
          ],
          "status": "passed",
          "detail": "Planner emitted the MCP tool request."
        },
        {
          "name": "mcp_tool",
          "tool": "get_controlled_context",
          "elapsedMs": 1,
          "firstTokenMs": null,
          "usage": null,
          "estimatedCostUsd": 0,
          "attempts": [
            {
              "attempt": 1,
              "status": 200,
              "elapsedMs": 1,
              "outcome": "success"
            }
          ],
          "status": "passed",
          "detail": "This MCP tool returns only the text supplied in the current run. It does not browse, infer telemetry, or access local files."
        },
        {
          "name": "answer",
          "model": "openai/gpt-oss-120b",
          "elapsedMs": 906,
          "firstTokenMs": 234,
          "usage": {
            "inputTokens": 303,
            "outputTokens": 179,
            "totalTokens": 482,
            "cachedTokens": 64
          },
          "estimatedCostUsd": 0.00015285,
          "attempts": [
            {
              "attempt": 1,
              "status": 200,
              "elapsedMs": 906,
              "outcome": "success"
            }
          ],
          "status": "passed",
          "detail": "Streaming completion finished with stop."
        }
      ],
      "reliabilityEvent": {
        "injected": true,
        "recovered": true,
        "status": 404,
        "elapsedMs": 172,
        "fallbackModel": "openai/gpt-oss-120b",
        "detail": "The deliberate invalid-model response was captured; the experiment continued with the configured route."
      }
    }
  ],
  "evaluation": {
    "model": {
      "id": "Qwen/Qwen3-30B-A3B-Instruct-2507",
      "name": "Qwen3-30B-A3B-Instruct-2507",
      "promptPricePerMillion": 0.1,
      "completionPricePerMillion": 0.3,
      "features": [
        "tools",
        "json_mode",
        "structured_outputs"
      ],
      "region": "eu-north1"
    },
    "scores": {
      "routeA": {
        "groundedness": 4,
        "actionability": 5,
        "clarity": 5,
        "note": "The response is grounded in the supplied context, specifically addressing the increase in payload size and token count due to shared context. It provides a concrete next step (using a token counter) and is clear and actionable for a developer."
      },
      "routeB": {
        "groundedness": 5,
        "actionability": 5,
        "clarity": 5,
        "note": "The response is fully grounded in the supplied context, correctly identifying that the shared context is being re-processed in full on every call, increasing token count and latency. It offers a clear, concrete next action (implementing a retrieval step for top-k relevant snippets) and is concise and understandable to a developer."
      },
      "winner": "B",
      "reason": "Route B provides a more complete and precise explanation of the root cause—reprocessing the full shared context on every call—while offering a stronger, more scalable solution (retrieval of relevant snippets) that directly addresses the performance issue. It is equally clear and actionable, but its diagnostic depth and architectural recommendation make it superior to Route A, which only suggests measuring tokens without proposing a structural fix."
    },
    "elapsedMs": 2816,
    "usage": {
      "inputTokens": 462,
      "outputTokens": 239,
      "totalTokens": 701,
      "cachedTokens": 128
    },
    "estimatedCostUsd": 0.0001179,
    "attempts": [
      {
        "attempt": 1,
        "status": 200,
        "elapsedMs": 2816,
        "outcome": "success"
      }
    ],
    "caveat": "One model-scored sample is directional evidence, not a benchmark or human evaluation."
  },
  "cacheProbe": {
    "model": {
      "id": "Qwen/Qwen3-30B-A3B-Instruct-2507",
      "name": "Qwen3-30B-A3B-Instruct-2507",
      "promptPricePerMillion": 0.1,
      "completionPricePerMillion": 0.3,
      "features": [
        "tools",
        "json_mode",
        "structured_outputs"
      ],
      "region": "eu-north1"
    },
    "elapsedMs": 1391,
    "firstTokenMs": 256,
    "usage": {
      "inputTokens": 242,
      "outputTokens": 99,
      "totalTokens": 341,
      "cachedTokens": 240
    },
    "estimatedCostUsd": 0.0000539,
    "attempts": [
      {
        "attempt": 1,
        "status": 200,
        "elapsedMs": 1391,
        "outcome": "success"
      }
    ],
    "cacheEvidence": "The repeated request returned 240 cached input tokens."
  },
  "summary": {
    "fastestObservedRoute": "A",
    "cheapestObservedRoute": "A",
    "evaluatorWinner": "B",
    "totalEstimatedCostUsd": 0.00046855,
    "reliabilityRecoveryObserved": true
  },
  "evidenceBoundary": {
    "observed": [
      "API model metadata",
      "provider response status",
      "end-to-end time",
      "streaming first-token time",
      "returned usage",
      "calculated token cost",
      "MCP tool result",
      "captured provider error"
    ],
    "notClaimed": [
      "platform-wide performance",
      "platform-wide cache benefit from one response",
      "production SLA",
      "human-rated answer quality"
    ]
  },
  "sources": [
    "https://docs.tokenfactory.nebius.com/ai-models-inference/function-calling",
    "https://docs.tokenfactory.nebius.com/ai-models-inference/observability",
    "https://docs.tokenfactory.nebius.com/api-reference/inference/create-chat-completion",
    "https://modelcontextprotocol.io/"
  ]
}
