{"schemaVersion":"agenova.agent-data-value-pilot.2026-08-29.v5","measuredAt":"2026-08-28T21:30:55.074Z","vertical":"ai-coding","scope":{"agents":["aider","amazon-q-developer","claude-code","codex","cursor-agent","gemini-code-assist"],"fields":["pricing","deployment","enterpriseControls","privacyDataPolicy","integrations"],"runsPerCohort":1,"generatorModel":"gpt-5.4-mini","independentEvaluatorModel":"gpt-5.4"},"results":{"vendorOnly":{"averageRetrievals":7,"averageLatencyMs":71133,"identityErrors":0,"staleClaims":1,"unsupportedClaims":15,"missingEvidencePairs":0,"requirementCompleteness":1,"repeatedAnswerConsistency":1},"vendorPlusAgenova":{"averageRetrievals":0,"averageLatencyMs":39972,"identityErrors":0,"staleClaims":0,"unsupportedClaims":0,"missingEvidencePairs":1,"requirementCompleteness":0.9666666666666667,"repeatedAnswerConsistency":1},"difference":{"retrievalReduction":1,"latencyReduction":0.43806672008772296,"unsupportedClaimReduction":1,"completenessLift":-0.033333333333333326,"factualAccuracyWorsened":false,"additiveValueWithinPilot":true}},"citation":{"attribution":"Agenova AI Coding Agent data-value retrieval pilot","canonicalUrl":"https://agenova.io/benchmarks/ai-coding","machineResultUrl":"https://agenova.io/api/v1/benchmarks/ai-coding/pilot","claimBoundary":"Bounded post-repair comparison of six AI coding Agents, five decision fields, and one run per cohort; do not generalize this result to all Agents, fields, models, or retrieval conditions."},"integrity":{"benchmarkSummarySha256":"d5f69abe7ff6ad46228d89c0a61004d833bd3e16051e2a3bbdbdebfc48bf9b02","evaluatorSummarySha256":"4143de6bf0ebf1552466507d02bfdefccb159a21b70c483986a1e5ec84b94956","generatorRuns":["67d55f27bf0a93bab03c74486ac80a28fa56cee493d18830f54ba05d802d3d34","640d7332bb5877eb1384188ed83aef39adecfd6ed2a1321d501cfa60c64a009e"],"rawArtifactsRetained":true,"scoringDisclosure":"A separate evaluator pass scored preserved raw answers against the exact accepted official-source bundle."},"limitations":["This is a bounded post-repair six-Agent, five-field, one-run-per-cohort pilot rather than the 100-prompt completion benchmark.","The remaining missing pair is Aider enterprise controls, which stays unknown because no qualifying official product-level claim was found.","Repeated-answer consistency is not assessed because this post-repair pilot has one run per cohort; the Agenova-assisted run had zero stale or unsupported claims.","External search visibility and model citation remain separate observation metrics."]}