{
  "version": "public-task-quality-v2",
  "published_date": "2026-09-25",
  "combined_score": null,
  "provenance": "Separate measurements of task relevance, plan preparation and coding. Development fixtures are not independently labeled real-world accuracy. These scores must not be combined into an end-to-end success rate.",
  "cards": [
    {
      "id": "task_relevance",
      "title": "Task relevance",
      "status": "Synthetic development test",
      "date": "2026-09-25",
      "model": "gpt-5.4-2026-03-05",
      "primary": {
        "label": "investigated cases passing all checks",
        "numerator": 14,
        "denominator": 14
      },
      "metrics": [
        {
          "label": "Real tasks retained",
          "numerator": 14,
          "denominator": 14
        },
        {
          "label": "Incorrect queue admissions",
          "numerator": 0,
          "denominator": 14
        },
        {
          "label": "Source-only baseline passing all checks",
          "numerator": 8,
          "denominator": 14
        }
      ],
      "limitations": [
        "14 agent-authored cases; live reviewer with frozen history, code and web retrieval. Planning is an invocation spy, not an executed plan.",
        "A real task may remain visible without belonging in our engineering queue. Human acceptance and execution approval remain separate."
      ]
    },
    {
      "id": "engineering_enrichment",
      "title": "Engineering enrichment",
      "status": "Synthetic development test",
      "date": "2026-09-25",
      "model": "gpt-5.4-2026-03-05 · low reasoning",
      "primary": {
        "label": "correct readiness decisions",
        "numerator": 24,
        "denominator": 24
      },
      "metrics": [
        {
          "label": "Completed cases",
          "numerator": 24,
          "denominator": 24
        },
        {
          "label": "False-ready decisions",
          "numerator": 0,
          "denominator": 12
        },
        {
          "label": "Automatic wording checks",
          "numerator": 23,
          "denominator": 24
        },
        {
          "label": "Plain-model readiness baseline",
          "numerator": 24,
          "denominator": 24
        },
        {
          "label": "Plain-model wording baseline",
          "numerator": 20,
          "denominator": 24
        },
        {
          "label": "Raw reviewer contrast checks",
          "numerator": 8,
          "denominator": 10
        }
      ],
      "limitations": [
        "24 agent-authored ambiguous tasks (16 original + 8 transfer). Readiness is not plan correctness. The separate raw reviewer scored 8/10, incorrectly rejecting two valid briefs.",
        "Same source evidence and per-call output cap; unequal total budget: up to 4 Actionairy calls versus 2 baseline calls. No claim of competitor superiority.",
        "Independently reviewed usable-plan accuracy and human correction time are not yet measured.",
        "Failed calls remain in the full denominator. Automatic wording checks are lexical, not semantic grading. No independent human usable-plan score is available."
      ]
    },
    {
      "id": "deepswe",
      "title": "Datacurve DeepSWE",
      "status": "Separate Compress Cloud run",
      "date": "2026-09-22",
      "model": "Archived Compress Cloud coding run",
      "primary": {
        "label": "full-cohort tasks passed",
        "numerator": 64,
        "denominator": 113
      },
      "metrics": [],
      "limitations": [
        "2 tasks remained ungradeable after bounded regrading. They receive no pass credit and stay in the full-cohort denominator.",
        "Sealed candidate patches and pinned verifier contexts. This is a separate coding evaluation, not an end-to-end Actionairy meeting-to-code test or a current-model leaderboard claim."
      ]
    }
  ],
  "capture_note": "Meeting extraction: independently labeled capture precision and recall are not yet available. Transcript-upload acceptance is a user-journey check, not an accuracy estimate.",
  "sources": [
    {
      "stage": "deepswe",
      "source_sha": null,
      "run_id": "6d1a182b-ffb3-418f-96d9-e30b0ecc2950",
      "report_sha256": "49bdc2775b52db68539de77b107cc71049589e6856d726cf5d465381cfa02a07"
    },
    {
      "stage": "deepswe_regrade",
      "source_sha": null,
      "run_id": "6d1a182b-ffb3-418f-96d9-e30b0ecc2950",
      "report_sha256": "190c9dff6fd4eda3e50d036f3623bb63047a44500ca279819fb072c15bc29ab7"
    },
    {
      "stage": "enrichment_original",
      "source_sha": "7979ebb8de981486f9de9379080945853a274aa3",
      "run_id": null,
      "report_sha256": "4d8e23f88a8127b19cbcd81541cbb257d17b0410f482ad051789da823acb94e1"
    },
    {
      "stage": "enrichment_transfer",
      "source_sha": "7979ebb8de981486f9de9379080945853a274aa3",
      "run_id": null,
      "report_sha256": "a22c9babf31fe339b5ce1c6efcd16041d5a5f6d273acc987342ab6787bb82b5a"
    },
    {
      "stage": "investigation",
      "source_sha": "7979ebb8de981486f9de9379080945853a274aa3",
      "run_id": null,
      "report_sha256": "d039fd648c0f059a6ca85241bdbb881ec7dd386de1f698b4efa350b8c1d0c9c8"
    },
    {
      "stage": "raw_review",
      "source_sha": "7979ebb8de981486f9de9379080945853a274aa3",
      "run_id": null,
      "report_sha256": "460f14648028294ec5366e45c89fec32bd1e157815bdd7e22bcb56e05ecfc682"
    }
  ]
}
