{
  "schema_version": 1,
  "title": "Long-video understanding: aggregate reference coverage",
  "scope": "A small development pilot with one reviewer and an assisted reference. Reference coverage measures inclusion of checked points, not overall factual accuracy, timestamp quality or performance as video duration increases.",
  "charts": [
    {
      "key": "workflow-coverage",
      "kind": "stacked",
      "rows": [
        {
          "label": "Agent Chat",
          "counts": [
            26,
            36,
            32
          ],
          "values": [
            27.659574468085108,
            38.297872340425535,
            34.04255319148936
          ],
          "denominator": 94
        },
        {
          "label": "Video Analysis",
          "counts": [
            80,
            13,
            1
          ],
          "values": [
            85.1063829787234,
            13.829787234042554,
            1.0638297872340425
          ],
          "denominator": 94
        }
      ],
      "unit": "percent",
      "title": "How much of the reference did each answer cover?",
      "series": [
        "Fully included",
        "Partly included",
        "Missing"
      ],
      "caption": "94 checked reference points per workflow, from the same completed, fully answerable requests. Fully included, partly included and missing are shown separately. Saved pilot outputs include revised answers."
    },
    {
      "key": "before-after",
      "kind": "grouped",
      "rows": [
        {
          "label": "Agent Chat",
          "counts": [
            20,
            20
          ],
          "values": [
            32.78688524590164,
            32.78688524590164
          ],
          "denominator": 61
        },
        {
          "label": "Video Analysis",
          "counts": [
            47,
            51
          ],
          "values": [
            77.04918032786885,
            83.60655737704919
          ],
          "denominator": 61
        }
      ],
      "unit": "percent",
      "title": "Coverage before and after improvements",
      "series": [
        "Earlier output",
        "Updated output"
      ],
      "caption": "The same requests and 61 reference-point checks per workflow before and after improvements. This is a development comparison using these examples, not an independent held-out evaluation."
    }
  ]
}