{"schema_version":1,"total":1,"offset":0,"limit":100,"observations":[{"id":"aa:21a0a2f6-bc72-40d2-80db-ea4e66b02b90:automationBenchPartialScore","benchmark_id":"aa-automationbench::1.0.6","subject":{"source_id":"21a0a2f6-bc72-40d2-80db-ea4e66b02b90","name":"GPT-6 Astra (Non-reasoning)","model_id":"gpt-6-astra::non-reasoning","variant":null,"harness":null},"value":0.6177224345856126,"unit":"fraction","basis":"measured","source":{"url":"https://artificialanalysis.ai/models/gpt-5-6-sol","retrieved_at":"2026-09-10T21:47:16.627Z","published_at":null,"sha256":"d921f7b255347b1091c5595adbbb5170ed6e9fe62250967a8586eb303ee99d2d","file":"ops/rebuild-2026-09/evidence/phase-04/sources/aa-29d2fef1679c.html.gz","locator":"model UUID 21a0a2f6-bc72-40d2-80db-ea4e66b02b90; automationBenchPartialScore"},"protocol":"Mean fraction of objectives completed; any guardrail violation or task error gives zero; AA methodology captured 2026-09-10; effort not in effort field (exact UUID retained)","comparison_key":null}],"divergences":[],"cell":null,"coverage":{"model":null,"benchmark":{"total_models":835,"available":153,"unknown":682,"not_tested":0,"not_published":0,"source_unreachable":0,"contested":0,"measured":153,"self_reported":0,"observations":155,"unmatched_observations":2}},"collection":{"benchmark_id":"aa-automationbench::1.0.6","status":"collected","source_url":"https://artificialanalysis.ai/models/gpt-5-6-sol","reason":"Exact AA UUID and field from the reviewed protocol snapshot; null does not prove whether AA ran the test."},"coverage_note":"Denominators are catalog model configurations × versioned registry entries. Basis counts count covered cells, not runs; measured and self_reported may overlap. Unmatched source identities remain in observations and never inflate catalog coverage."}