{
  "schemaVersion": "1.1",
  "updated": "2026-09-19",
  "completedRuns": 0,
  "runs": [
    {
      "testId": "ATG-DOC-CHATGPT-001",
      "protocolId": "DOC-GROUNDING-V1",
      "tool": "chatgpt",
      "provider": "OpenAI",
      "category": "document-analysis",
      "testDate": null,
      "modelVersion": null,
      "plan": null,
      "platform": null,
      "promptTask": "Use the published DOC-GROUNDING-V1 pack exactly as written.",
      "expectedBehavior": "Answer known facts exactly; identify the date contradiction; preserve the accessibility qualifier; do not invent a vendor; trace claims to the supplied documents.",
      "actualResult": null,
      "factualErrors": null,
      "unsupportedClaims": null,
      "citationCoverage": null,
      "citationCorrectness": null,
      "correctionEffort": null,
      "completionSuccess": null,
      "majorLimitation": null,
      "evidenceReference": null,
      "notes": "Awaiting a real run and evidence.",
      "status": "awaiting-evidence",
      "runConditions": null,
      "firstResponseSha256": null,
      "evidenceSha256": null,
      "scoringVersion": null,
      "scoredBy": null,
      "scoredAt": null
    },
    {
      "testId": "ATG-DOC-CLAUDE-001",
      "protocolId": "DOC-GROUNDING-V1",
      "tool": "claude",
      "provider": "Anthropic",
      "category": "document-analysis",
      "testDate": null,
      "modelVersion": null,
      "plan": null,
      "platform": null,
      "promptTask": "Use the published DOC-GROUNDING-V1 pack exactly as written.",
      "expectedBehavior": "Answer known facts exactly; identify the date contradiction; preserve the accessibility qualifier; do not invent a vendor; trace claims to the supplied documents.",
      "actualResult": null,
      "factualErrors": null,
      "unsupportedClaims": null,
      "citationCoverage": null,
      "citationCorrectness": null,
      "correctionEffort": null,
      "completionSuccess": null,
      "majorLimitation": null,
      "evidenceReference": null,
      "notes": "Awaiting a real run and evidence.",
      "status": "awaiting-evidence",
      "runConditions": null,
      "firstResponseSha256": null,
      "evidenceSha256": null,
      "scoringVersion": null,
      "scoredBy": null,
      "scoredAt": null
    },
    {
      "testId": "ATG-DOC-GEMINI-001",
      "protocolId": "DOC-GROUNDING-V1",
      "tool": "gemini",
      "provider": "Google",
      "category": "document-analysis",
      "testDate": null,
      "modelVersion": null,
      "plan": null,
      "platform": null,
      "promptTask": "Use the published DOC-GROUNDING-V1 pack exactly as written.",
      "expectedBehavior": "Answer known facts exactly; identify the date contradiction; preserve the accessibility qualifier; do not invent a vendor; trace claims to the supplied documents.",
      "actualResult": null,
      "factualErrors": null,
      "unsupportedClaims": null,
      "citationCoverage": null,
      "citationCorrectness": null,
      "correctionEffort": null,
      "completionSuccess": null,
      "majorLimitation": null,
      "evidenceReference": null,
      "notes": "Awaiting a real run and evidence.",
      "status": "awaiting-evidence",
      "runConditions": null,
      "firstResponseSha256": null,
      "evidenceSha256": null,
      "scoringVersion": null,
      "scoredBy": null,
      "scoredAt": null
    },
    {
      "testId": "ATG-DOC-PERPLEXITY-001",
      "protocolId": "DOC-GROUNDING-V1",
      "tool": "perplexity-ai",
      "provider": "Perplexity",
      "category": "document-analysis",
      "testDate": null,
      "modelVersion": null,
      "plan": null,
      "platform": null,
      "promptTask": "Use the published DOC-GROUNDING-V1 pack exactly as written.",
      "expectedBehavior": "Answer known facts exactly; identify the date contradiction; preserve the accessibility qualifier; do not invent a vendor; trace claims to the supplied documents.",
      "actualResult": null,
      "factualErrors": null,
      "unsupportedClaims": null,
      "citationCoverage": null,
      "citationCorrectness": null,
      "correctionEffort": null,
      "completionSuccess": null,
      "majorLimitation": null,
      "evidenceReference": null,
      "notes": "Awaiting a real run and evidence.",
      "status": "awaiting-evidence",
      "runConditions": null,
      "firstResponseSha256": null,
      "evidenceSha256": null,
      "scoringVersion": null,
      "scoredBy": null,
      "scoredAt": null
    },
    {
      "testId": "ATG-DOC-COPILOT-001",
      "protocolId": "DOC-GROUNDING-V1",
      "tool": "microsoft-copilot",
      "provider": "Microsoft",
      "category": "document-analysis",
      "testDate": null,
      "modelVersion": null,
      "plan": null,
      "platform": null,
      "promptTask": "Use the published DOC-GROUNDING-V1 pack exactly as written.",
      "expectedBehavior": "Answer known facts exactly; identify the date contradiction; preserve the accessibility qualifier; do not invent a vendor; trace claims to the supplied documents.",
      "actualResult": null,
      "factualErrors": null,
      "unsupportedClaims": null,
      "citationCoverage": null,
      "citationCorrectness": null,
      "correctionEffort": null,
      "completionSuccess": null,
      "majorLimitation": null,
      "evidenceReference": null,
      "notes": "Awaiting a real run and evidence.",
      "status": "awaiting-evidence",
      "runConditions": null,
      "firstResponseSha256": null,
      "evidenceSha256": null,
      "scoringVersion": null,
      "scoredBy": null,
      "scoredAt": null
    },
    {
      "testId": "ATG-IMG-FIREFLY-001",
      "protocolId": "IMAGE-CONSTRAINT-V1",
      "tool": "adobe-firefly",
      "provider": "Adobe",
      "category": "image-generation",
      "testDate": null,
      "modelVersion": null,
      "plan": null,
      "platform": null,
      "promptTask": "Use the published IMAGE-CONSTRAINT-V1 prompt exactly as written.",
      "expectedBehavior": "Follow the countable object, text, layout, prohibited-element and output-condition constraints without inventing a subjective overall score.",
      "actualResult": null,
      "factualErrors": null,
      "unsupportedClaims": null,
      "citationCoverage": null,
      "citationCorrectness": null,
      "correctionEffort": null,
      "completionSuccess": null,
      "majorLimitation": null,
      "evidenceReference": null,
      "notes": "Awaiting a real run and evidence.",
      "status": "awaiting-evidence",
      "runConditions": null,
      "firstResponseSha256": null,
      "evidenceSha256": null,
      "scoringVersion": null,
      "scoredBy": null,
      "scoredAt": null
    }
  ]
}
