{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://ysymyth.github.io/The-Second-Half/"], "content_sha256": "211dcd740a14ba0b10babff55cc7cb2ee86244e82ab3362d885829ee2d164e14", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/yao-second-half.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-001-the-second-half.raw.txt", "primary_url": "https://ysymyth.github.io/The-Second-Half/"}, "exemplar_id": "awesome_evals::ae-001-the-second-half", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-001-the-second-half::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-001-the-second-half", "source_section": "⭐ Must-read starter set (read these first)", "source_title": "The Second Half", "source_type": "web_article", "source_url": "https://ysymyth.github.io/The-Second-Half/"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://eugeneyan.com/writing/eval-process/"], "content_sha256": "9b432b0da627bdda285f2fe9acb204ed196470da3c096e89f96e550999fea84d", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-002-an-llm-as-judge-won-t-save-the-product-fixing-yo.raw.txt", "primary_url": "https://eugeneyan.com/writing/eval-process/"}, "exemplar_id": "awesome_evals::ae-002-an-llm-as-judge-won-t-save-the-product-fixing-yo", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-002-an-llm-as-judge-won-t-save-the-product-fixing-yo::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-002-an-llm-as-judge-won-t-save-the-product-fixing-yo", "source_section": "⭐ Must-read starter set (read these first)", "source_title": "An LLM-as-Judge Won't Save the Product, Fixing Your Process Will", "source_type": "web_article", "source_url": "https://eugeneyan.com/writing/eval-process/"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://leehanchung.github.io/blogs/2026/06/13/hidden-technical-debt-agent-evaluation-infra/"], "content_sha256": "0c84dd759fb4cd17d700488033f3ff82b01da2f20fb197182868451e77e008d0", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/leehanchung-hidden-technical-debt-agent-runtime.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-003-hidden-technical-debt-agent-evaluation-infrastru.raw.txt", "primary_url": "https://leehanchung.github.io/blogs/2026/06/13/hidden-technical-debt-agent-evaluation-infra/"}, "exemplar_id": "awesome_evals::ae-003-hidden-technical-debt-agent-evaluation-infrastru", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-003-hidden-technical-debt-agent-evaluation-infrastru::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-003-hidden-technical-debt-agent-evaluation-infrastru", "source_section": "⭐ Must-read starter set (read these first)", "source_title": "Hidden Technical Debt: Agent Evaluation Infrastructure", "source_type": "blog", "source_url": "https://leehanchung.github.io/blogs/2026/06/13/hidden-technical-debt-agent-evaluation-infra/"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://hamel.dev/blog/posts/evals-faq/"], "content_sha256": "b5d5398f91d39542cc52d6c11bc38dfb86e4da2d06fcb73e69add860602d8e02", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-004-llm-evals-faq.raw.txt", "primary_url": "https://hamel.dev/blog/posts/evals-faq/"}, "exemplar_id": "awesome_evals::ae-004-llm-evals-faq", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-004-llm-evals-faq::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-004-llm-evals-faq", "source_section": "⭐ Must-read starter set (read these first)", "source_title": "LLM Evals FAQ", "source_type": "blog", "source_url": "https://hamel.dev/blog/posts/evals-faq/"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://www.jasonwei.net/blog/asymmetry-of-verification-and-verifiers-law"], "content_sha256": "be6f54cb997e1d6095c362a8d722e30de68819787535cf8385441c8c8ea43492", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-005-asymmetry-of-verification-and-verifier-s-law.raw.txt", "primary_url": "https://www.jasonwei.net/blog/asymmetry-of-verification-and-verifiers-law"}, "exemplar_id": "awesome_evals::ae-005-asymmetry-of-verification-and-verifier-s-law", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-005-asymmetry-of-verification-and-verifier-s-law::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-005-asymmetry-of-verification-and-verifier-s-law", "source_section": "⭐ Must-read starter set (read these first)", "source_title": "Asymmetry of Verification and Verifier's Law", "source_type": "blog", "source_url": "https://www.jasonwei.net/blog/asymmetry-of-verification-and-verifiers-law"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://www.anthropic.com/engineering/demystifying-evals-for-ai-agents"], "content_sha256": "59ea5d13bccd08ddfc548b26d0ce4591c4047c07592472ed35c011de38c5f258", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/vanishing-gradients-agents-evals.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-006-demystifying-evals-for-ai-agents.raw.txt", "primary_url": "https://www.anthropic.com/engineering/demystifying-evals-for-ai-agents"}, "exemplar_id": "awesome_evals::ae-006-demystifying-evals-for-ai-agents", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-006-demystifying-evals-for-ai-agents::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-006-demystifying-evals-for-ai-agents", "source_section": "⭐ Must-read starter set (read these first)", "source_title": "Demystifying Evals for AI Agents", "source_type": "web_article", "source_url": "https://www.anthropic.com/engineering/demystifying-evals-for-ai-agents"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://ofir.io/How-to-Build-Good-Language-Modeling-Benchmarks/"], "content_sha256": "cb4d4263c57f9558283b48d61549aa99a8931294452efa8cbe2860c1b4c0b7cd", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-007-how-to-build-good-language-modeling-benchmarks.raw.txt", "primary_url": "https://ofir.io/How-to-Build-Good-Language-Modeling-Benchmarks/"}, "exemplar_id": "awesome_evals::ae-007-how-to-build-good-language-modeling-benchmarks", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-007-how-to-build-good-language-modeling-benchmarks::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-007-how-to-build-good-language-modeling-benchmarks", "source_section": "⭐ Must-read starter set (read these first)", "source_title": "How to Build Good Language Modeling Benchmarks", "source_type": "web_article", "source_url": "https://ofir.io/How-to-Build-Good-Language-Modeling-Benchmarks/"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://arxiv.org/abs/2407.01502"], "content_sha256": "a74a66bddeea47fd5b656b5bd316858bb9baf7b951491316ec773e834c43f53c", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-008-ai-agents-that-matter.raw.txt", "primary_url": "https://arxiv.org/abs/2407.01502"}, "exemplar_id": "awesome_evals::ae-008-ai-agents-that-matter", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-008-ai-agents-that-matter::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-008-ai-agents-that-matter", "source_section": "⭐ Must-read starter set (read these first)", "source_title": "AI Agents That Matter", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2407.01502"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://www.interconnects.ai/p/building-on-evaluation-quicksand"], "content_sha256": "b9885f095cf3cc54596b20530d30dbfae608ada2f47dec0aaa19cd751e883fb6", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-009-building-on-evaluation-quicksand.raw.txt", "primary_url": "https://www.interconnects.ai/p/building-on-evaluation-quicksand"}, "exemplar_id": "awesome_evals::ae-009-building-on-evaluation-quicksand", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-009-building-on-evaluation-quicksand::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-009-building-on-evaluation-quicksand", "source_section": "⭐ Must-read starter set (read these first)", "source_title": "Building on Evaluation Quicksand", "source_type": "blog", "source_url": "https://www.interconnects.ai/p/building-on-evaluation-quicksand"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://arxiv.org/abs/2404.12272"], "content_sha256": "2a98cac4c82e1b225ab4c6468262c9245923882e276c6f518b0142cc02ecae5a", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-010-who-validates-the-validators-evalgen.raw.txt", "primary_url": "https://arxiv.org/abs/2404.12272"}, "exemplar_id": "awesome_evals::ae-010-who-validates-the-validators-evalgen", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-010-who-validates-the-validators-evalgen::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-010-who-validates-the-validators-evalgen", "source_section": "⭐ Must-read starter set (read these first)", "source_title": "Who Validates the Validators? (EvalGen)", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2404.12272"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://florianbrand.com/posts/benches-2026"], "content_sha256": "fc8d7716215d0260805cab35106c2f3c3fd1384fe70eaa946cbe3d80a5d2eebe", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/florian-brand-prime-intellect-llm-benchmarks-era-of-agents.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-011-benches-2026-llm-benchmarks-in-the-era-of-agents.raw.txt", "primary_url": "https://florianbrand.com/posts/benches-2026"}, "exemplar_id": "awesome_evals::ae-011-benches-2026-llm-benchmarks-in-the-era-of-agents", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-011-benches-2026-llm-benchmarks-in-the-era-of-agents::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-011-benches-2026-llm-benchmarks-in-the-era-of-agents", "source_section": "⭐ Must-read starter set (read these first)", "source_title": "Benches 2026 — \"LLM benchmarks in the era of agents\"", "source_type": "web_article", "source_url": "https://florianbrand.com/posts/benches-2026"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://openai.com/index/trustworthy-third-party-evaluations-foundations/"], "content_sha256": "442f66e4c9dc7a6e09364da71f1c14ca0fdf85402fe2c01e903a33ef1c4659ad", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-012-a-shared-playbook-for-trustworthy-third-party-ev.raw.txt", "primary_url": "https://openai.com/index/trustworthy-third-party-evaluations-foundations/"}, "exemplar_id": "awesome_evals::ae-012-a-shared-playbook-for-trustworthy-third-party-ev", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-012-a-shared-playbook-for-trustworthy-third-party-ev::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-012-a-shared-playbook-for-trustworthy-third-party-ev", "source_section": "⭐ Must-read starter set (read these first)", "source_title": "A Shared Playbook for Trustworthy Third-Party Evaluations", "source_type": "docs_or_book", "source_url": "https://openai.com/index/trustworthy-third-party-evaluations-foundations/"}
{"eval_domain": "eval_motivation", "evidence": {"all_urls": ["https://ysymyth.github.io/The-Second-Half/"], "content_sha256": "211dcd740a14ba0b10babff55cc7cb2ee86244e82ab3362d885829ee2d164e14", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/yao-second-half.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-013-the-second-half.raw.txt", "primary_url": "https://ysymyth.github.io/The-Second-Half/"}, "exemplar_id": "awesome_evals::ae-013-the-second-half", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-013-the-second-half::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclear_success_criterion", "source_id": "ae-013-the-second-half", "source_section": "1 · Why we need evals", "source_title": "The Second Half", "source_type": "web_article", "source_url": "https://ysymyth.github.io/The-Second-Half/"}
{"eval_domain": "eval_motivation", "evidence": {"all_urls": ["https://eugeneyan.com/writing/eval-process/"], "content_sha256": "b08e2853cdf61563b2217567d7226d7c6a1c40cb6dfdc4185affbc4e9231cbea", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-014-an-llm-as-judge-won-t-save-the-product-fixing-yo.raw.txt", "primary_url": "https://eugeneyan.com/writing/eval-process/"}, "exemplar_id": "awesome_evals::ae-014-an-llm-as-judge-won-t-save-the-product-fixing-yo", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-014-an-llm-as-judge-won-t-save-the-product-fixing-yo::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclear_success_criterion", "source_id": "ae-014-an-llm-as-judge-won-t-save-the-product-fixing-yo", "source_section": "1 · Why we need evals", "source_title": "An LLM-as-Judge Won't Save the Product, Fixing Your Process Will", "source_type": "web_article", "source_url": "https://eugeneyan.com/writing/eval-process/"}
{"eval_domain": "eval_motivation", "evidence": {"all_urls": ["https://hamel.dev/blog/posts/evals/"], "content_sha256": "6083d93b8cd419ee5b9bf5400bd4c2de09461095d39a172f1d4939ed7b4a1950", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-015-your-ai-product-needs-evals.raw.txt", "primary_url": "https://hamel.dev/blog/posts/evals/"}, "exemplar_id": "awesome_evals::ae-015-your-ai-product-needs-evals", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-015-your-ai-product-needs-evals::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclear_success_criterion", "source_id": "ae-015-your-ai-product-needs-evals", "source_section": "1 · Why we need evals", "source_title": "Your AI Product Needs Evals", "source_type": "blog", "source_url": "https://hamel.dev/blog/posts/evals/"}
{"eval_domain": "eval_motivation", "evidence": {"all_urls": ["https://hamel.dev/blog/posts/field-guide/"], "content_sha256": "7ce00cfbe0dea2e01003ed2f138b7d48c317796161c58f030b6020db255f4c3b", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-pod-vg50-field-guide.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-016-a-field-guide-to-rapidly-improving-ai-products.raw.txt", "primary_url": "https://hamel.dev/blog/posts/field-guide/"}, "exemplar_id": "awesome_evals::ae-016-a-field-guide-to-rapidly-improving-ai-products", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-016-a-field-guide-to-rapidly-improving-ai-products::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclear_success_criterion", "source_id": "ae-016-a-field-guide-to-rapidly-improving-ai-products", "source_section": "1 · Why we need evals", "source_title": "A Field Guide to Rapidly Improving AI Products", "source_type": "blog", "source_url": "https://hamel.dev/blog/posts/field-guide/"}
{"eval_domain": "eval_motivation", "evidence": {"all_urls": ["https://www.sh-reya.com/blog/in-defense-ai-evals/"], "content_sha256": "2f37bf6b0ad6b9c317b0d5b083418b3a88de027d3bdb265a3398e058c61640de", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-017-in-defense-of-ai-evals-for-everyone.raw.txt", "primary_url": "https://www.sh-reya.com/blog/in-defense-ai-evals/"}, "exemplar_id": "awesome_evals::ae-017-in-defense-of-ai-evals-for-everyone", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-017-in-defense-of-ai-evals-for-everyone::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclear_success_criterion", "source_id": "ae-017-in-defense-of-ai-evals-for-everyone", "source_section": "1 · Why we need evals", "source_title": "In Defense of AI Evals, for Everyone", "source_type": "blog", "source_url": "https://www.sh-reya.com/blog/in-defense-ai-evals/"}
{"eval_domain": "eval_motivation", "evidence": {"all_urls": ["https://applied-llms.org/", "https://www.oreilly.com/radar/what-we-learned-from-a-year-of-building-with-llms-part-ii/"], "content_sha256": "e8b4979dd187bfda59f3a57b7692da0ef99b7d5c8791d52a6674a7e6e817a90d", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-018-what-we-learned-from-a-year-of-building-with-llm.raw.txt", "primary_url": "https://applied-llms.org/"}, "exemplar_id": "awesome_evals::ae-018-what-we-learned-from-a-year-of-building-with-llm", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-018-what-we-learned-from-a-year-of-building-with-llm::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclear_success_criterion", "source_id": "ae-018-what-we-learned-from-a-year-of-building-with-llm", "source_section": "1 · Why we need evals", "source_title": "What We Learned from a Year of Building with LLMs", "source_type": "web_article", "source_url": "https://applied-llms.org/"}
{"eval_domain": "eval_motivation", "evidence": {"all_urls": ["https://www.interconnects.ai/p/evals-are-marketing"], "content_sha256": "81d5d69d5dae4f0d01c4847454f4d10c091cbef50697b1d20bb6eed5375968cf", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-019-big-tech-s-llm-evals-are-just-marketing.raw.txt", "primary_url": "https://www.interconnects.ai/p/evals-are-marketing"}, "exemplar_id": "awesome_evals::ae-019-big-tech-s-llm-evals-are-just-marketing", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-019-big-tech-s-llm-evals-are-just-marketing::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclear_success_criterion", "source_id": "ae-019-big-tech-s-llm-evals-are-just-marketing", "source_section": "1 · Why we need evals", "source_title": "Big Tech's LLM Evals Are Just Marketing", "source_type": "blog", "source_url": "https://www.interconnects.ai/p/evals-are-marketing"}
{"eval_domain": "eval_motivation", "evidence": {"all_urls": ["https://huyenchip.com/2025/01/16/ai-engineering-pitfalls.html"], "content_sha256": "4ca5bbd7e2f6dc1f367046187ab1bcc4a59a58c5f9aa11f0a2d237238c830ba6", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-020-ai-engineering-pitfalls.raw.txt", "primary_url": "https://huyenchip.com/2025/01/16/ai-engineering-pitfalls.html"}, "exemplar_id": "awesome_evals::ae-020-ai-engineering-pitfalls", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-020-ai-engineering-pitfalls::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclear_success_criterion", "source_id": "ae-020-ai-engineering-pitfalls", "source_section": "1 · Why we need evals", "source_title": "AI Engineering pitfalls", "source_type": "web_article", "source_url": "https://huyenchip.com/2025/01/16/ai-engineering-pitfalls.html"}
{"eval_domain": "eval_motivation", "evidence": {"all_urls": ["https://www.oreilly.com/radar/evals-are-not-all-you-need/"], "content_sha256": "a1b9a080abd9e102912af7d460cd9694989db50ef3b84749c28351fa5e6012f2", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-021-evals-are-not-all-you-need.raw.txt", "primary_url": "https://www.oreilly.com/radar/evals-are-not-all-you-need/"}, "exemplar_id": "awesome_evals::ae-021-evals-are-not-all-you-need", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-021-evals-are-not-all-you-need::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclear_success_criterion", "source_id": "ae-021-evals-are-not-all-you-need", "source_section": "1 · Why we need evals", "source_title": "Evals Are NOT All You Need", "source_type": "web_article", "source_url": "https://www.oreilly.com/radar/evals-are-not-all-you-need/"}
{"eval_domain": "eval_motivation", "evidence": {"all_urls": ["https://www.lennysnewsletter.com/p/why-ai-evals-are-the-hottest-new-skill"], "content_sha256": "a350cfc5d90110de1388a2c93065e33643c7076d3a37940cc3522d7850e7c220", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-022-why-ai-evals-are-the-hottest-new-skill-for-produ.raw.txt", "primary_url": "https://www.lennysnewsletter.com/p/why-ai-evals-are-the-hottest-new-skill"}, "exemplar_id": "awesome_evals::ae-022-why-ai-evals-are-the-hottest-new-skill-for-produ", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-022-why-ai-evals-are-the-hottest-new-skill-for-produ::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclear_success_criterion", "source_id": "ae-022-why-ai-evals-are-the-hottest-new-skill-for-produ", "source_section": "1 · Why we need evals", "source_title": "Why AI evals are the hottest new skill for product builders", "source_type": "newsletter", "source_url": "https://www.lennysnewsletter.com/p/why-ai-evals-are-the-hottest-new-skill"}
{"eval_domain": "eval_motivation", "evidence": {"all_urls": ["https://openai.com/index/evals-drive-next-chapter-of-ai/"], "content_sha256": "8c78f8a8986123a7b89e0ef97b482a0bb625c8f743695b2f466700d0c33b0c98", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-023-how-evals-drive-the-next-chapter-in-ai-for-busin.raw.txt", "primary_url": "https://openai.com/index/evals-drive-next-chapter-of-ai/"}, "exemplar_id": "awesome_evals::ae-023-how-evals-drive-the-next-chapter-in-ai-for-busin", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-023-how-evals-drive-the-next-chapter-in-ai-for-busin::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclear_success_criterion", "source_id": "ae-023-how-evals-drive-the-next-chapter-in-ai-for-busin", "source_section": "1 · Why we need evals", "source_title": "How evals drive the next chapter in AI for businesses", "source_type": "web_article", "source_url": "https://openai.com/index/evals-drive-next-chapter-of-ai/"}
{"eval_domain": "eval_motivation", "evidence": {"all_urls": ["https://www.lennysnewsletter.com/p/beyond-vibe-checks-a-pms-complete"], "content_sha256": "adc8b3738c1323790bd7e7edb0aa0a4d28ede9e5027a9b2fa28552e22de66269", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/aman-khan-beyond-vibe-checks-pm-guide-evals.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-024-beyond-vibe-checks-a-pm-s-complete-guide-to-eval.raw.txt", "primary_url": "https://www.lennysnewsletter.com/p/beyond-vibe-checks-a-pms-complete"}, "exemplar_id": "awesome_evals::ae-024-beyond-vibe-checks-a-pm-s-complete-guide-to-eval", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-024-beyond-vibe-checks-a-pm-s-complete-guide-to-eval::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclear_success_criterion", "source_id": "ae-024-beyond-vibe-checks-a-pm-s-complete-guide-to-eval", "source_section": "1 · Why we need evals", "source_title": "Beyond vibe checks: A PM's complete guide to evals", "source_type": "newsletter", "source_url": "https://www.lennysnewsletter.com/p/beyond-vibe-checks-a-pms-complete"}
{"eval_domain": "eval_motivation", "evidence": {"all_urls": ["https://newsletter.pragmaticengineer.com/p/evals"], "content_sha256": "398ba7259b1827e0aad73d9f2441a07ea8adc978d343c541c5ea99b2cf277adf", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/pragmatic-engineer-llm-evals-guide.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-025-a-pragmatic-guide-to-llm-evals-for-devs.raw.txt", "primary_url": "https://newsletter.pragmaticengineer.com/p/evals"}, "exemplar_id": "awesome_evals::ae-025-a-pragmatic-guide-to-llm-evals-for-devs", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-025-a-pragmatic-guide-to-llm-evals-for-devs::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclear_success_criterion", "source_id": "ae-025-a-pragmatic-guide-to-llm-evals-for-devs", "source_section": "1 · Why we need evals", "source_title": "A pragmatic guide to LLM evals for devs", "source_type": "newsletter", "source_url": "https://newsletter.pragmaticengineer.com/p/evals"}
{"eval_domain": "eval_motivation", "evidence": {"all_urls": ["https://openai.com/index/deployment-simulation/"], "content_sha256": "1c4e765138fefa384a61328801cb79c2bd42642f8de19c03d7bc27dae925bf51", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-026-predicting-model-behavior-before-release-by-simu.raw.txt", "primary_url": "https://openai.com/index/deployment-simulation/"}, "exemplar_id": "awesome_evals::ae-026-predicting-model-behavior-before-release-by-simu", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-026-predicting-model-behavior-before-release-by-simu::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclear_success_criterion", "source_id": "ae-026-predicting-model-behavior-before-release-by-simu", "source_section": "1 · Why we need evals", "source_title": "Predicting model behavior before release by simulating deployment (Deployment Simulation)", "source_type": "web_article", "source_url": "https://openai.com/index/deployment-simulation/"}
{"eval_domain": "eval_motivation", "evidence": {"all_urls": ["https://x.com/gdb/status/1733553161884127435"], "content_sha256": null, "http_status": null, "local_note_path": null, "local_raw_path": null, "primary_url": "https://x.com/gdb/status/1733553161884127435"}, "exemplar_id": "awesome_evals::ae-027-evals-are-surprisingly-often-all-you-need", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-027-evals-are-surprisingly-often-all-you-need::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclear_success_criterion", "source_id": "ae-027-evals-are-surprisingly-often-all-you-need", "source_section": "1 · Why we need evals", "source_title": "evals are surprisingly often all you need", "source_type": "web_article", "source_url": "https://x.com/gdb/status/1733553161884127435"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://www.jasonwei.net/blog/asymmetry-of-verification-and-verifiers-law"], "content_sha256": "be6f54cb997e1d6095c362a8d722e30de68819787535cf8385441c8c8ea43492", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-028-asymmetry-of-verification-and-verifier-s-law.raw.txt", "primary_url": "https://www.jasonwei.net/blog/asymmetry-of-verification-and-verifiers-law"}, "exemplar_id": "awesome_evals::ae-028-asymmetry-of-verification-and-verifier-s-law", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-028-asymmetry-of-verification-and-verifier-s-law::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-028-asymmetry-of-verification-and-verifier-s-law", "source_section": "2 · \"If you can eval it, you have built it\" — eval ⇄ capability ⇄ RL environment", "source_title": "Asymmetry of Verification and Verifier's Law", "source_type": "blog", "source_url": "https://www.jasonwei.net/blog/asymmetry-of-verification-and-verifiers-law"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://leehanchung.github.io/blogs/2026/03/21/rl-environments-for-llm-agents/"], "content_sha256": "066a8cb3086350a324ccd4d36c1cc212ec168dbfe1db7c186759de70098a1e10", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-029-a-taxonomy-of-rl-environments-for-llm-agents.raw.txt", "primary_url": "https://leehanchung.github.io/blogs/2026/03/21/rl-environments-for-llm-agents/"}, "exemplar_id": "awesome_evals::ae-029-a-taxonomy-of-rl-environments-for-llm-agents", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-029-a-taxonomy-of-rl-environments-for-llm-agents::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-029-a-taxonomy-of-rl-environments-for-llm-agents", "source_section": "2 · \"If you can eval it, you have built it\" — eval ⇄ capability ⇄ RL environment", "source_title": "A Taxonomy of RL Environments for LLM Agents", "source_type": "blog", "source_url": "https://leehanchung.github.io/blogs/2026/03/21/rl-environments-for-llm-agents/"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://muratbuffalo.blogspot.com/2026/06/acm-cais-conference-on-ai-and-agentic.html"], "content_sha256": "2b1eab44957e019167a49304099ab4d5700712cd2a57d6f88df48a96c772b60e", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-030-the-life-cycle-of-an-rl-environment.raw.txt", "primary_url": "https://muratbuffalo.blogspot.com/2026/06/acm-cais-conference-on-ai-and-agentic.html"}, "exemplar_id": "awesome_evals::ae-030-the-life-cycle-of-an-rl-environment", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-030-the-life-cycle-of-an-rl-environment::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-030-the-life-cycle-of-an-rl-environment", "source_section": "2 · \"If you can eval it, you have built it\" — eval ⇄ capability ⇄ RL environment", "source_title": "The Life Cycle of an RL Environment", "source_type": "blog", "source_url": "https://muratbuffalo.blogspot.com/2026/06/acm-cais-conference-on-ai-and-agentic.html"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://storage.googleapis.com/deepmind-media/Era-of-Experience%20/The%20Era%20of%20Experience%20Paper.pdf"], "content_sha256": "af7c19261a953b194acd69d6c594f1ee64f8f5175a0df000212d8fb3cc409a63", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-031-welcome-to-the-era-of-experience.raw.txt", "primary_url": "https://storage.googleapis.com/deepmind-media/Era-of-Experience%20/The%20Era%20of%20Experience%20Paper.pdf"}, "exemplar_id": "awesome_evals::ae-031-welcome-to-the-era-of-experience", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-031-welcome-to-the-era-of-experience::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-031-welcome-to-the-era-of-experience", "source_section": "2 · \"If you can eval it, you have built it\" — eval ⇄ capability ⇄ RL environment", "source_title": "Welcome to the Era of Experience", "source_type": "paper_or_pdf", "source_url": "https://storage.googleapis.com/deepmind-media/Era-of-Experience%20/The%20Era%20of%20Experience%20Paper.pdf"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://rlhfbook.com/c/16-evaluation"], "content_sha256": "20f54a4a48bea184c60f73e0d56cf5482f7bb0c2033875b6dbb475531a3b2583", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-032-rlhf-book-ch-16-evaluation.raw.txt", "primary_url": "https://rlhfbook.com/c/16-evaluation"}, "exemplar_id": "awesome_evals::ae-032-rlhf-book-ch-16-evaluation", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-032-rlhf-book-ch-16-evaluation::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-032-rlhf-book-ch-16-evaluation", "source_section": "2 · \"If you can eval it, you have built it\" — eval ⇄ capability ⇄ RL environment", "source_title": "RLHF Book, Ch. 16 — Evaluation", "source_type": "docs_or_book", "source_url": "https://rlhfbook.com/c/16-evaluation"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://www.interconnects.ai/p/what-comes-next-with-reinforcement"], "content_sha256": "18660f7c802931d0f69c8989922deb1cbdaba4031643e6f6f8e4c6fa784179a1", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/papers/playing-atari-with-deep-reinforcement-learning.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-033-what-comes-next-with-reinforcement-learning.raw.txt", "primary_url": "https://www.interconnects.ai/p/what-comes-next-with-reinforcement"}, "exemplar_id": "awesome_evals::ae-033-what-comes-next-with-reinforcement-learning", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-033-what-comes-next-with-reinforcement-learning::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-033-what-comes-next-with-reinforcement-learning", "source_section": "2 · \"If you can eval it, you have built it\" — eval ⇄ capability ⇄ RL environment", "source_title": "What Comes Next with Reinforcement Learning", "source_type": "blog", "source_url": "https://www.interconnects.ai/p/what-comes-next-with-reinforcement"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://github.com/PrimeIntellect-ai/verifiers"], "content_sha256": "120d283ec24119b3b3b128edbd58203efe95ee1601e126ff1a95c29e8c89d348", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-034-verifiers.raw.txt", "primary_url": "https://github.com/PrimeIntellect-ai/verifiers"}, "exemplar_id": "awesome_evals::ae-034-verifiers", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-034-verifiers::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-034-verifiers", "source_section": "2 · \"If you can eval it, you have built it\" — eval ⇄ capability ⇄ RL environment", "source_title": "verifiers", "source_type": "repository_or_docs", "source_url": "https://github.com/PrimeIntellect-ai/verifiers"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2501.12948"], "content_sha256": "0166b468d8d8650065cb9b8a72017b25eb51970fc948c2c59f5d4d66fe088171", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/papers/playing-atari-with-deep-reinforcement-learning.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-035-deepseek-r1-incentivizing-reasoning-capability-i.raw.txt", "primary_url": "https://arxiv.org/abs/2501.12948"}, "exemplar_id": "awesome_evals::ae-035-deepseek-r1-incentivizing-reasoning-capability-i", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-035-deepseek-r1-incentivizing-reasoning-capability-i::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-035-deepseek-r1-incentivizing-reasoning-capability-i", "source_section": "2 · \"If you can eval it, you have built it\" — eval ⇄ capability ⇄ RL environment", "source_title": "DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2501.12948"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2411.15124"], "content_sha256": "608d288e2b22dd53d93ddf43e39220214987ea988d082bd0f58fd30b23f0deb3", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/jason-wei-successful-language-model-evals.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-036-t-lu-3-pushing-frontiers-in-open-language-model.raw.txt", "primary_url": "https://arxiv.org/abs/2411.15124"}, "exemplar_id": "awesome_evals::ae-036-t-lu-3-pushing-frontiers-in-open-language-model", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-036-t-lu-3-pushing-frontiers-in-open-language-model::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-036-t-lu-3-pushing-frontiers-in-open-language-model", "source_section": "2 · \"If you can eval it, you have built it\" — eval ⇄ capability ⇄ RL environment", "source_title": "Tülu 3: Pushing Frontiers in Open Language Model Post-Training", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2411.15124"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://www.anthropic.com/research/emergent-misalignment-reward-hacking"], "content_sha256": "2b43ff6518f92c8404b48abdcb2ba4d38d696b0cc3b9b514407f5b4ff763a4ec", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/natural-emergent-misalignment-reward-hacking-production-rl.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-037-natural-emergent-misalignment-from-reward-hackin.raw.txt", "primary_url": "https://www.anthropic.com/research/emergent-misalignment-reward-hacking"}, "exemplar_id": "awesome_evals::ae-037-natural-emergent-misalignment-from-reward-hackin", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-037-natural-emergent-misalignment-from-reward-hackin::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-037-natural-emergent-misalignment-from-reward-hackin", "source_section": "2 · \"If you can eval it, you have built it\" — eval ⇄ capability ⇄ RL environment", "source_title": "Natural Emergent Misalignment from Reward Hacking in Production RL", "source_type": "web_article", "source_url": "https://www.anthropic.com/research/emergent-misalignment-reward-hacking"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://www.primeintellect.ai/blog/environments"], "content_sha256": "3a7642ae07ed957f61c6ecd8034515ca89ff74ca3dd93359a37f9c685002d9b9", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-brown-rl-environments-at-scale.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-038-environments-hub-a-community-hub-to-scale-rl-to.raw.txt", "primary_url": "https://www.primeintellect.ai/blog/environments"}, "exemplar_id": "awesome_evals::ae-038-environments-hub-a-community-hub-to-scale-rl-to", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-038-environments-hub-a-community-hub-to-scale-rl-to::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-038-environments-hub-a-community-hub-to-scale-rl-to", "source_section": "2 · \"If you can eval it, you have built it\" — eval ⇄ capability ⇄ RL environment", "source_title": "Environments Hub: A Community Hub To Scale RL To Open AGI", "source_type": "blog", "source_url": "https://www.primeintellect.ai/blog/environments"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://www.mechanize.work/blog/how-to-fully-automate-software-engineering/"], "content_sha256": "b67d2e4aae274b04bba1efa3a5ab3065f14598d748ca7ce096215217c15d9242", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-039-how-to-fully-automate-software-engineering.raw.txt", "primary_url": "https://www.mechanize.work/blog/how-to-fully-automate-software-engineering/"}, "exemplar_id": "awesome_evals::ae-039-how-to-fully-automate-software-engineering", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-039-how-to-fully-automate-software-engineering::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-039-how-to-fully-automate-software-engineering", "source_section": "2 · \"If you can eval it, you have built it\" — eval ⇄ capability ⇄ RL environment", "source_title": "How to fully automate software engineering", "source_type": "blog", "source_url": "https://www.mechanize.work/blog/how-to-fully-automate-software-engineering/"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://www.mechanize.work/blog/cheap-rl-tasks-will-waste-compute/"], "content_sha256": "271dccf94d271a6a222bac5c8936cf42ae7763f0448cdf3d3f47d7fa482b58c4", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-040-cheap-rl-tasks-will-waste-compute.raw.txt", "primary_url": "https://www.mechanize.work/blog/cheap-rl-tasks-will-waste-compute/"}, "exemplar_id": "awesome_evals::ae-040-cheap-rl-tasks-will-waste-compute", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-040-cheap-rl-tasks-will-waste-compute::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-040-cheap-rl-tasks-will-waste-compute", "source_section": "2 · \"If you can eval it, you have built it\" — eval ⇄ capability ⇄ RL environment", "source_title": "Cheap RL tasks will waste compute", "source_type": "blog", "source_url": "https://www.mechanize.work/blog/cheap-rl-tasks-will-waste-compute/"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://epoch.ai/gradient-updates/state-of-rl-envs"], "content_sha256": "090462c797763c8be18e23b01c6285997b2d8ef9c70362909707530eb4065793", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/papers/playing-atari-with-deep-reinforcement-learning.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-041-an-faq-on-reinforcement-learning-environments.raw.txt", "primary_url": "https://epoch.ai/gradient-updates/state-of-rl-envs"}, "exemplar_id": "awesome_evals::ae-041-an-faq-on-reinforcement-learning-environments", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-041-an-faq-on-reinforcement-learning-environments::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-041-an-faq-on-reinforcement-learning-environments", "source_section": "2 · \"If you can eval it, you have built it\" — eval ⇄ capability ⇄ RL environment", "source_title": "An FAQ on Reinforcement Learning Environments", "source_type": "web_article", "source_url": "https://epoch.ai/gradient-updates/state-of-rl-envs"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://newsletter.semianalysis.com/p/rl-environments-and-rl-for-science"], "content_sha256": "b50e2819061359867ed522647aeb77c5d8c804e8abf8fccdd6690f3a683b8c57", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/mast-why-multi-agent-llm-systems-fail.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-042-rl-environments-and-rl-for-science-data-foundrie.raw.txt", "primary_url": "https://newsletter.semianalysis.com/p/rl-environments-and-rl-for-science"}, "exemplar_id": "awesome_evals::ae-042-rl-environments-and-rl-for-science-data-foundrie", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-042-rl-environments-and-rl-for-science-data-foundrie::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-042-rl-environments-and-rl-for-science-data-foundrie", "source_section": "2 · \"If you can eval it, you have built it\" — eval ⇄ capability ⇄ RL environment", "source_title": "RL Environments and RL for Science: Data Foundries and Multi-Agent Architectures", "source_type": "newsletter", "source_url": "https://newsletter.semianalysis.com/p/rl-environments-and-rl-for-science"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://github.com/harbor-framework/terminal-bench"], "content_sha256": "86101ef2cf5623a17effe698d87d61ddc32ea4aabe3914d3aa03e141dfda3a2c", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-043-terminal-bench-benchmarking-agents-on-hard-reali.raw.txt", "primary_url": "https://github.com/harbor-framework/terminal-bench"}, "exemplar_id": "awesome_evals::ae-043-terminal-bench-benchmarking-agents-on-hard-reali", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-043-terminal-bench-benchmarking-agents-on-hard-reali::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-043-terminal-bench-benchmarking-agents-on-hard-reali", "source_section": "2 · \"If you can eval it, you have built it\" — eval ⇄ capability ⇄ RL environment", "source_title": "Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces", "source_type": "repository_or_docs", "source_url": "https://github.com/harbor-framework/terminal-bench"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://github.com/sierra-research/tau2-bench"], "content_sha256": "b5654a3b95286c0b722b6940e52dcda06762ece148d8df56e85f9a58e6fc5088", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-044-tau2-bench-bench-a-benchmark-for-tool-agent-user.raw.txt", "primary_url": "https://github.com/sierra-research/tau2-bench"}, "exemplar_id": "awesome_evals::ae-044-tau2-bench-bench-a-benchmark-for-tool-agent-user", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-044-tau2-bench-bench-a-benchmark-for-tool-agent-user::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-044-tau2-bench-bench-a-benchmark-for-tool-agent-user", "source_section": "2 · \"If you can eval it, you have built it\" — eval ⇄ capability ⇄ RL environment", "source_title": "tau2-bench (τ²-Bench): A Benchmark for Tool-Agent-User Interaction in Real-World Domains", "source_type": "repository_or_docs", "source_url": "https://github.com/sierra-research/tau2-bench"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://leehanchung.github.io/blogs/2026/05/08/hidden-technical-debt-agent-harness/"], "content_sha256": "82bab3c182f4749115098e4e8adb24c9e873e83b32b22525592ea2a6eecaa6c3", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/leehanchung-hidden-technical-debt-agent-runtime.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-045-hidden-technical-debt-agent-harness.raw.txt", "primary_url": "https://leehanchung.github.io/blogs/2026/05/08/hidden-technical-debt-agent-harness/"}, "exemplar_id": "awesome_evals::ae-045-hidden-technical-debt-agent-harness", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-045-hidden-technical-debt-agent-harness::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-045-hidden-technical-debt-agent-harness", "source_section": "3 · The model / harness / skill decomposition", "source_title": "Hidden Technical Debt: Agent Harness", "source_type": "blog", "source_url": "https://leehanchung.github.io/blogs/2026/05/08/hidden-technical-debt-agent-harness/"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://leehanchung.github.io/blogs/"], "content_sha256": "feb7c83720c58480af79bad2784ed943ef402e15052184af4ea4681d7e5c9a5c", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/leehanchung-hidden-technical-debt-agent-runtime.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-046-hidden-technical-debt-series-index.raw.txt", "primary_url": "https://leehanchung.github.io/blogs/"}, "exemplar_id": "awesome_evals::ae-046-hidden-technical-debt-series-index", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-046-hidden-technical-debt-series-index::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-046-hidden-technical-debt-series-index", "source_section": "3 · The model / harness / skill decomposition", "source_title": "Hidden Technical Debt series (index)", "source_type": "blog", "source_url": "https://leehanchung.github.io/blogs/"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://metr.org/blog/2025-03-19-measuring-ai-ability-to-complete-long-tasks/"], "content_sha256": "b3e67dedc0e8c78eab25f1f94034a6959f378e16be2126fdbc619bee781dbb8e", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-047-measuring-ai-ability-to-complete-long-tasks.raw.txt", "primary_url": "https://metr.org/blog/2025-03-19-measuring-ai-ability-to-complete-long-tasks/"}, "exemplar_id": "awesome_evals::ae-047-measuring-ai-ability-to-complete-long-tasks", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-047-measuring-ai-ability-to-complete-long-tasks::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-047-measuring-ai-ability-to-complete-long-tasks", "source_section": "3 · The model / harness / skill decomposition", "source_title": "Measuring AI Ability to Complete Long Tasks", "source_type": "blog", "source_url": "https://metr.org/blog/2025-03-19-measuring-ai-ability-to-complete-long-tasks/"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://www.turingpost.com/p/nathanlambert"], "content_sha256": "ae261462d45aa59da5d49c5aff7776bf66b7c58ad50e2e313340cd866b6b7729", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-048-turing-post-interview-open-models-won-t-catch-up.raw.txt", "primary_url": "https://www.turingpost.com/p/nathanlambert"}, "exemplar_id": "awesome_evals::ae-048-turing-post-interview-open-models-won-t-catch-up", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-048-turing-post-interview-open-models-won-t-catch-up::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-048-turing-post-interview-open-models-won-t-catch-up", "source_section": "3 · The model / harness / skill decomposition", "source_title": "Turing Post interview (\"Open Models Won't Catch Up\")", "source_type": "blog", "source_url": "https://www.turingpost.com/p/nathanlambert"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://florianbrand.com/posts/benches-2026", "https://www.youtube.com/watch?v=kmTMc-fVSXw"], "content_sha256": "fc8d7716215d0260805cab35106c2f3c3fd1384fe70eaa946cbe3d80a5d2eebe", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-049-quo-vadis-llm-benchmarks.raw.txt", "primary_url": "https://florianbrand.com/posts/benches-2026"}, "exemplar_id": "awesome_evals::ae-049-quo-vadis-llm-benchmarks", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-049-quo-vadis-llm-benchmarks::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-049-quo-vadis-llm-benchmarks", "source_section": "3 · The model / harness / skill decomposition", "source_title": "Quo vadis, LLM benchmarks?", "source_type": "web_article", "source_url": "https://florianbrand.com/posts/benches-2026"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://leehanchung.github.io/talks/2025/04/23/the-model-is-the-product/"], "content_sha256": "da98c7012d76059c648a438af889a3e23f92e9a31b53b4f3b050b0c1bd37ed1d", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-lee-model-is-the-product.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-050-the-model-is-the-product.raw.txt", "primary_url": "https://leehanchung.github.io/talks/2025/04/23/the-model-is-the-product/"}, "exemplar_id": "awesome_evals::ae-050-the-model-is-the-product", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-050-the-model-is-the-product::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-050-the-model-is-the-product", "source_section": "3 · The model / harness / skill decomposition", "source_title": "The Model is the Product", "source_type": "web_article", "source_url": "https://leehanchung.github.io/talks/2025/04/23/the-model-is-the-product/"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=EEw2PpL-_NM"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-lee-model-is-the-product.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=EEw2PpL-_NM"}, "exemplar_id": "awesome_evals::ae-051-the-model-is-not-the-product", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-051-the-model-is-not-the-product::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-051-the-model-is-not-the-product", "source_section": "3 · The model / harness / skill decomposition", "source_title": "The Model is Not the Product", "source_type": "talk_video", "source_url": "https://www.youtube.com/watch?v=EEw2PpL-_NM"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://simonwillison.net/2025/May/22/tools-in-a-loop/"], "content_sha256": "df8b6cc6587ea72a46702facd7b3a844910bff4441b4165df9012df0d2f49bd5", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/anthropic-writing-tools-for-agents.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-052-agents-are-models-using-tools-in-a-loop.raw.txt", "primary_url": "https://simonwillison.net/2025/May/22/tools-in-a-loop/"}, "exemplar_id": "awesome_evals::ae-052-agents-are-models-using-tools-in-a-loop", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-052-agents-are-models-using-tools-in-a-loop::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-052-agents-are-models-using-tools-in-a-loop", "source_section": "3 · The model / harness / skill decomposition", "source_title": "Agents are models using tools in a loop", "source_type": "web_article", "source_url": "https://simonwillison.net/2025/May/22/tools-in-a-loop/"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://openai.com/index/harness-engineering/"], "content_sha256": "c8c5abb4fa8b87d6ecb2705f3e2f2a587c8bc29d86d30e6ce8c4a3505ccf8627", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-053-harness-engineering-leveraging-codex-in-an-agent.raw.txt", "primary_url": "https://openai.com/index/harness-engineering/"}, "exemplar_id": "awesome_evals::ae-053-harness-engineering-leveraging-codex-in-an-agent", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-053-harness-engineering-leveraging-codex-in-an-agent::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-053-harness-engineering-leveraging-codex-in-an-agent", "source_section": "3 · The model / harness / skill decomposition", "source_title": "Harness engineering: leveraging Codex in an agent-first world", "source_type": "web_article", "source_url": "https://openai.com/index/harness-engineering/"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://www.anthropic.com/engineering/equipping-agents-for-the-real-world-with-agent-skills"], "content_sha256": "628fd795d4babeb906966f46c121212ed39349349d8160fd665d24ffe8fc7df7", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/vercel-agents-md-outperforms-skills.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-054-equipping-agents-for-the-real-world-with-agent-s.raw.txt", "primary_url": "https://www.anthropic.com/engineering/equipping-agents-for-the-real-world-with-agent-skills"}, "exemplar_id": "awesome_evals::ae-054-equipping-agents-for-the-real-world-with-agent-s", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-054-equipping-agents-for-the-real-world-with-agent-s::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-054-equipping-agents-for-the-real-world-with-agent-s", "source_section": "3 · The model / harness / skill decomposition", "source_title": "Equipping agents for the real world with Agent Skills", "source_type": "web_article", "source_url": "https://www.anthropic.com/engineering/equipping-agents-for-the-real-world-with-agent-skills"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://www.anthropic.com/engineering/effective-context-engineering-for-ai-agents"], "content_sha256": "b4ae6b5269833d884203ffee69144bff4eb5bfa0bafca3a5de59132dd5c6bbbd", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-055-effective-context-engineering-for-ai-agents.raw.txt", "primary_url": "https://www.anthropic.com/engineering/effective-context-engineering-for-ai-agents"}, "exemplar_id": "awesome_evals::ae-055-effective-context-engineering-for-ai-agents", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-055-effective-context-engineering-for-ai-agents::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-055-effective-context-engineering-for-ai-agents", "source_section": "3 · The model / harness / skill decomposition", "source_title": "Effective context engineering for AI agents", "source_type": "web_article", "source_url": "https://www.anthropic.com/engineering/effective-context-engineering-for-ai-agents"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://www.anthropic.com/engineering/writing-tools-for-agents"], "content_sha256": "3c2de66140551d40a57cb1ab38ea1812bc19260473c8484bc0dacab3ab75711a", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/anthropic-writing-tools-for-agents.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-056-writing-effective-tools-for-agents-with-agents.raw.txt", "primary_url": "https://www.anthropic.com/engineering/writing-tools-for-agents"}, "exemplar_id": "awesome_evals::ae-056-writing-effective-tools-for-agents-with-agents", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-056-writing-effective-tools-for-agents-with-agents::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-056-writing-effective-tools-for-agents-with-agents", "source_section": "3 · The model / harness / skill decomposition", "source_title": "Writing effective tools for agents — with agents", "source_type": "web_article", "source_url": "https://www.anthropic.com/engineering/writing-tools-for-agents"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://blog.thepete.net/blog/2025/12/10/same-model-different-results-why-coding-agents-arent-interchangeable/"], "content_sha256": "5ff566fd14e1ce4f1b016c3b83765a6de3e14df125b140b3b1622ff95e4c1963", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-057-same-model-different-results-why-coding-agents-a.raw.txt", "primary_url": "https://blog.thepete.net/blog/2025/12/10/same-model-different-results-why-coding-agents-arent-interchangeable/"}, "exemplar_id": "awesome_evals::ae-057-same-model-different-results-why-coding-agents-a", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-057-same-model-different-results-why-coding-agents-a::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-057-same-model-different-results-why-coding-agents-a", "source_section": "3 · The model / harness / skill decomposition", "source_title": "Same Model, Different Results: Why Coding Agents Aren't Interchangeable", "source_type": "blog", "source_url": "https://blog.thepete.net/blog/2025/12/10/same-model-different-results-why-coding-agents-arent-interchangeable/"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://hal.cs.princeton.edu/"], "content_sha256": "3846d6ab680ab4b8664169b0cbd83421344a6b6286205f86380b0a1f4b349d2a", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-058-holistic-agent-leaderboard-hal.raw.txt", "primary_url": "https://hal.cs.princeton.edu/"}, "exemplar_id": "awesome_evals::ae-058-holistic-agent-leaderboard-hal", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-058-holistic-agent-leaderboard-hal::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-058-holistic-agent-leaderboard-hal", "source_section": "3 · The model / harness / skill decomposition", "source_title": "Holistic Agent Leaderboard (HAL)", "source_type": "web_article", "source_url": "https://hal.cs.princeton.edu/"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://www.oreilly.com/radar/agent-harness-engineering/"], "content_sha256": "7b74b1d3828bb96450fc2b3fd863e5b341a30d8f6d5c7d71719a2edae538f296", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-059-agent-harness-engineering.raw.txt", "primary_url": "https://www.oreilly.com/radar/agent-harness-engineering/"}, "exemplar_id": "awesome_evals::ae-059-agent-harness-engineering", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-059-agent-harness-engineering::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-059-agent-harness-engineering", "source_section": "3 · The model / harness / skill decomposition", "source_title": "Agent Harness Engineering", "source_type": "web_article", "source_url": "https://www.oreilly.com/radar/agent-harness-engineering/"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://www.interconnects.ai/p/the-next-phase-of-open-models"], "content_sha256": "da81065268df58f08729537ea671a8ffbd15b8269dfc4d392edf06987d8f2419", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/papers/red-teaming-language-models-with-language-models.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-060-what-comes-next-with-open-models-weights-tools-h.raw.txt", "primary_url": "https://www.interconnects.ai/p/the-next-phase-of-open-models"}, "exemplar_id": "awesome_evals::ae-060-what-comes-next-with-open-models-weights-tools-h", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-060-what-comes-next-with-open-models-weights-tools-h::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-060-what-comes-next-with-open-models-weights-tools-h", "source_section": "3 · The model / harness / skill decomposition", "source_title": "What comes next with open models (weights / tools / harness decomposition)", "source_type": "blog", "source_url": "https://www.interconnects.ai/p/the-next-phase-of-open-models"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://leehanchung.github.io/blogs/2026/06/13/hidden-technical-debt-agent-evaluation-infra/"], "content_sha256": "0c84dd759fb4cd17d700488033f3ff82b01da2f20fb197182868451e77e008d0", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/leehanchung-hidden-technical-debt-agent-runtime.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-061-hidden-technical-debt-agent-evaluation-infrastru.raw.txt", "primary_url": "https://leehanchung.github.io/blogs/2026/06/13/hidden-technical-debt-agent-evaluation-infra/"}, "exemplar_id": "awesome_evals::ae-061-hidden-technical-debt-agent-evaluation-infrastru", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-061-hidden-technical-debt-agent-evaluation-infrastru::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-061-hidden-technical-debt-agent-evaluation-infrastru", "source_section": "4 · Observability & the output / eval space (the surfaces you can grade)", "source_title": "Hidden Technical Debt: Agent Evaluation Infrastructure", "source_type": "blog", "source_url": "https://leehanchung.github.io/blogs/2026/06/13/hidden-technical-debt-agent-evaluation-infra/"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://www.braintrust.dev/blog/three-pillars-ai-observability"], "content_sha256": "0130840f40c5485535e3e1c49f5bf87294d0eeb6e7318699195e6323d0794584", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-062-the-three-pillars-of-ai-observability.raw.txt", "primary_url": "https://www.braintrust.dev/blog/three-pillars-ai-observability"}, "exemplar_id": "awesome_evals::ae-062-the-three-pillars-of-ai-observability", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-062-the-three-pillars-of-ai-observability::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-062-the-three-pillars-of-ai-observability", "source_section": "4 · Observability & the output / eval space (the surfaces you can grade)", "source_title": "The Three Pillars of AI Observability", "source_type": "blog", "source_url": "https://www.braintrust.dev/blog/three-pillars-ai-observability"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://arize.com/docs/ax/evaluate/evaluators/trace-and-session-evals/trace-level-evaluations/agent-trajectory-evaluations"], "content_sha256": "27e8cd4f1e97c60212035e9dce04e770ba1c891b5bbbe30d249c0d421ededd41", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/agentrewardbench-evaluating-automatic-evaluations-web-agent-.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-063-agent-trajectory-evaluations.raw.txt", "primary_url": "https://arize.com/docs/ax/evaluate/evaluators/trace-and-session-evals/trace-level-evaluations/agent-trajectory-evaluations"}, "exemplar_id": "awesome_evals::ae-063-agent-trajectory-evaluations", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-063-agent-trajectory-evaluations::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-063-agent-trajectory-evaluations", "source_section": "4 · Observability & the output / eval space (the surfaces you can grade)", "source_title": "Agent Trajectory Evaluations", "source_type": "docs_or_book", "source_url": "https://arize.com/docs/ax/evaluate/evaluators/trace-and-session-evals/trace-level-evaluations/agent-trajectory-evaluations"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://galileo.ai/blog/ai-agent-metrics"], "content_sha256": "4001f491e3157cbf2839c72c8036a9849738822ecb4bd4d1320272b48187fb7f", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/opik-evaluate-agent-trajectory.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-064-ai-agent-metrics-how-elite-teams-evaluate.raw.txt", "primary_url": "https://galileo.ai/blog/ai-agent-metrics"}, "exemplar_id": "awesome_evals::ae-064-ai-agent-metrics-how-elite-teams-evaluate", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-064-ai-agent-metrics-how-elite-teams-evaluate::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-064-ai-agent-metrics-how-elite-teams-evaluate", "source_section": "4 · Observability & the output / eval space (the surfaces you can grade)", "source_title": "AI Agent Metrics: How Elite Teams Evaluate", "source_type": "blog", "source_url": "https://galileo.ai/blog/ai-agent-metrics"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://github.com/Arize-ai/openinference/blob/main/spec/semantic_conventions.md"], "content_sha256": "d703927c37762d39268e593f37c9fcd228ca171e4dd62054e6b7909357da39bf", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-065-openinference-semantic-conventions.md", "primary_url": "https://github.com/Arize-ai/openinference/blob/main/spec/semantic_conventions.md"}, "exemplar_id": "awesome_evals::ae-065-openinference-semantic-conventions", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-065-openinference-semantic-conventions::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-065-openinference-semantic-conventions", "source_section": "4 · Observability & the output / eval space (the surfaces you can grade)", "source_title": "OpenInference semantic conventions", "source_type": "repository_or_docs", "source_url": "https://github.com/Arize-ai/openinference/blob/main/spec/semantic_conventions.md"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://docs.langchain.com/langsmith/evaluation", "https://docs.langchain.com/langsmith/trajectory-evals"], "content_sha256": "b5004e48455bb6fcb22c5fc737830a5d63289cfabe84280d0eb21168eccdd8d4", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-066-langsmith-evaluation-trajectory-evals.raw.txt", "primary_url": "https://docs.langchain.com/langsmith/evaluation"}, "exemplar_id": "awesome_evals::ae-066-langsmith-evaluation-trajectory-evals", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-066-langsmith-evaluation-trajectory-evals::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-066-langsmith-evaluation-trajectory-evals", "source_section": "4 · Observability & the output / eval space (the surfaces you can grade)", "source_title": "LangSmith Evaluation / Trajectory evals", "source_type": "docs_or_book", "source_url": "https://docs.langchain.com/langsmith/evaluation"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://github.com/open-telemetry/semantic-conventions-genai"], "content_sha256": "4a820be3a0a40ea3271e791320c126516f629af0beaef80cd0bde5162ff66429", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-067-opentelemetry-genai-semantic-conventions-agent-f.raw.txt", "primary_url": "https://github.com/open-telemetry/semantic-conventions-genai"}, "exemplar_id": "awesome_evals::ae-067-opentelemetry-genai-semantic-conventions-agent-f", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-067-opentelemetry-genai-semantic-conventions-agent-f::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-067-opentelemetry-genai-semantic-conventions-agent-f", "source_section": "4 · Observability & the output / eval space (the surfaces you can grade)", "source_title": "OpenTelemetry GenAI Semantic Conventions (agent & framework spans)", "source_type": "repository_or_docs", "source_url": "https://github.com/open-telemetry/semantic-conventions-genai"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://opentelemetry.io/docs/specs/semconv/gen-ai/gen-ai-agent-spans/"], "content_sha256": "cd400ffb83545fe0f2d39b7c6ec5adfed163d9423b408303214d7f3be07072f1", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-068-semantic-conventions-for-genai-agent-and-framewo.raw.txt", "primary_url": "https://opentelemetry.io/docs/specs/semconv/gen-ai/gen-ai-agent-spans/"}, "exemplar_id": "awesome_evals::ae-068-semantic-conventions-for-genai-agent-and-framewo", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-068-semantic-conventions-for-genai-agent-and-framewo::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-068-semantic-conventions-for-genai-agent-and-framewo", "source_section": "4 · Observability & the output / eval space (the surfaces you can grade)", "source_title": "Semantic Conventions for GenAI agent and framework spans", "source_type": "docs_or_book", "source_url": "https://opentelemetry.io/docs/specs/semconv/gen-ai/gen-ai-agent-spans/"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://opentelemetry.io/blog/2026/genai-observability/"], "content_sha256": "5933bf129610655098ffe87f4496a74182c5ea3393aa4e672ee77fe708a34b23", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-069-inside-the-llm-call-genai-observability-with-ope.raw.txt", "primary_url": "https://opentelemetry.io/blog/2026/genai-observability/"}, "exemplar_id": "awesome_evals::ae-069-inside-the-llm-call-genai-observability-with-ope", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-069-inside-the-llm-call-genai-observability-with-ope::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-069-inside-the-llm-call-genai-observability-with-ope", "source_section": "4 · Observability & the output / eval space (the surfaces you can grade)", "source_title": "Inside the LLM Call: GenAI Observability with OpenTelemetry", "source_type": "blog", "source_url": "https://opentelemetry.io/blog/2026/genai-observability/"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://docs.wandb.ai/weave"], "content_sha256": "b737a9cb3ccf6999ad8980c81aa303e6cf7b2675f8dcaba635a71fa077107cd4", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-070-w-b-weave-tracing-evaluation-toolkit.raw.txt", "primary_url": "https://docs.wandb.ai/weave"}, "exemplar_id": "awesome_evals::ae-070-w-b-weave-tracing-evaluation-toolkit", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-070-w-b-weave-tracing-evaluation-toolkit::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-070-w-b-weave-tracing-evaluation-toolkit", "source_section": "4 · Observability & the output / eval space (the surfaces you can grade)", "source_title": "W&B Weave — tracing & evaluation toolkit", "source_type": "docs_or_book", "source_url": "https://docs.wandb.ai/weave"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://laminar.sh/"], "content_sha256": "7a07d97e415b9f5495599f73eea40ac1d14b620faf5be3697fb4e4ba21c50c89", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/hf-agents-course-bonus-unit2-observability-evaluation.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-071-laminar-open-source-observability-for-ai-agents.raw.txt", "primary_url": "https://laminar.sh/"}, "exemplar_id": "awesome_evals::ae-071-laminar-open-source-observability-for-ai-agents", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-071-laminar-open-source-observability-for-ai-agents::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-071-laminar-open-source-observability-for-ai-agents", "source_section": "4 · Observability & the output / eval space (the surfaces you can grade)", "source_title": "Laminar — open-source observability for AI agents", "source_type": "web_article", "source_url": "https://laminar.sh/"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://github.com/UKGovernmentBEIS/inspect_ai", "https://inspect.aisi.org.uk/"], "content_sha256": "36f94cd023966fd3eadd5e36e57056c9240bf7beb7ec5a801ff683880eb56496", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-072-inspect-ai.raw.txt", "primary_url": "https://github.com/UKGovernmentBEIS/inspect_ai"}, "exemplar_id": "awesome_evals::ae-072-inspect-ai", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-072-inspect-ai::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-072-inspect-ai", "source_section": "5a · Eval frameworks & harnesses (code-first test-runners)", "source_title": "Inspect AI", "source_type": "repository_or_docs", "source_url": "https://github.com/UKGovernmentBEIS/inspect_ai"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://github.com/UKGovernmentBEIS/inspect_evals"], "content_sha256": "c26349f4f1da9c1324690868bc8590b0f7e803566364dcee0d6c17b03f3bedcc", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-073-inspect-evals.raw.txt", "primary_url": "https://github.com/UKGovernmentBEIS/inspect_evals"}, "exemplar_id": "awesome_evals::ae-073-inspect-evals", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-073-inspect-evals::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-073-inspect-evals", "source_section": "5a · Eval frameworks & harnesses (code-first test-runners)", "source_title": "inspect_evals", "source_type": "repository_or_docs", "source_url": "https://github.com/UKGovernmentBEIS/inspect_evals"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://github.com/EleutherAI/lm-evaluation-harness"], "content_sha256": "12e85f3de7be0931231408220b7930a19f1d653d2368636dba2dabacf5d4a665", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-074-lm-evaluation-harness.raw.txt", "primary_url": "https://github.com/EleutherAI/lm-evaluation-harness"}, "exemplar_id": "awesome_evals::ae-074-lm-evaluation-harness", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-074-lm-evaluation-harness::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-074-lm-evaluation-harness", "source_section": "5a · Eval frameworks & harnesses (code-first test-runners)", "source_title": "lm-evaluation-harness", "source_type": "repository_or_docs", "source_url": "https://github.com/EleutherAI/lm-evaluation-harness"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://github.com/allenai/olmes"], "content_sha256": "206ceebd3a44c3020b218ac6705e054b68d5e7d3cec93a6a2972c6753f36ae8b", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-075-olmes.raw.txt", "primary_url": "https://github.com/allenai/olmes"}, "exemplar_id": "awesome_evals::ae-075-olmes", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-075-olmes::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-075-olmes", "source_section": "5a · Eval frameworks & harnesses (code-first test-runners)", "source_title": "OLMES", "source_type": "repository_or_docs", "source_url": "https://github.com/allenai/olmes"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://github.com/benchflow-ai/benchflow", "https://benchflow.ai"], "content_sha256": "4e95d90950ce815ebe0b35665f308e14acbda95d9d793b2699c99d0a08bcb1ab", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-076-benchflow.raw.txt", "primary_url": "https://github.com/benchflow-ai/benchflow"}, "exemplar_id": "awesome_evals::ae-076-benchflow", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-076-benchflow::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-076-benchflow", "source_section": "5a · Eval frameworks & harnesses (code-first test-runners)", "source_title": "BenchFlow", "source_type": "repository_or_docs", "source_url": "https://github.com/benchflow-ai/benchflow"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://github.com/huggingface/lighteval"], "content_sha256": "7e8476b66f44c8fdc2b7da2b2cd1f124cf61a0dc8683f0a78c121050a3e333b8", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-077-lighteval.raw.txt", "primary_url": "https://github.com/huggingface/lighteval"}, "exemplar_id": "awesome_evals::ae-077-lighteval", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-077-lighteval::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-077-lighteval", "source_section": "5a · Eval frameworks & harnesses (code-first test-runners)", "source_title": "lighteval", "source_type": "repository_or_docs", "source_url": "https://github.com/huggingface/lighteval"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://github.com/groq/openbench"], "content_sha256": "4e0f378a1752c6c881786b488f1fc2b383866b23e1a71da5861ac3c0abc732ce", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-078-openbench.raw.txt", "primary_url": "https://github.com/groq/openbench"}, "exemplar_id": "awesome_evals::ae-078-openbench", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-078-openbench::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-078-openbench", "source_section": "5a · Eval frameworks & harnesses (code-first test-runners)", "source_title": "OpenBench", "source_type": "repository_or_docs", "source_url": "https://github.com/groq/openbench"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://github.com/openai/simple-evals"], "content_sha256": "b55c06cb3e6a509a20a5bf060ec8a570b04c01259477c2fa01ef8f87968dbc6f", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-079-simple-evals.raw.txt", "primary_url": "https://github.com/openai/simple-evals"}, "exemplar_id": "awesome_evals::ae-079-simple-evals", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-079-simple-evals::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-079-simple-evals", "source_section": "5a · Eval frameworks & harnesses (code-first test-runners)", "source_title": "simple-evals", "source_type": "repository_or_docs", "source_url": "https://github.com/openai/simple-evals"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://github.com/openai/evals", "https://developers.openai.com/api/docs/guides/evaluation-best-practices"], "content_sha256": "5931ce26961f019f4e85379d120c8e43f06379dd5ad51713114c5d921bce73ed", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/macro-evals-agentic-systems-openai-cookbook.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-080-openai-evals.raw.txt", "primary_url": "https://github.com/openai/evals"}, "exemplar_id": "awesome_evals::ae-080-openai-evals", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-080-openai-evals::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-080-openai-evals", "source_section": "5a · Eval frameworks & harnesses (code-first test-runners)", "source_title": "OpenAI Evals", "source_type": "repository_or_docs", "source_url": "https://github.com/openai/evals"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://github.com/promptfoo/promptfoo"], "content_sha256": "39e5d541c92970d8b75b49e701ae6fe75695b142321a38112cc14c88e8ad0361", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-081-promptfoo.raw.txt", "primary_url": "https://github.com/promptfoo/promptfoo"}, "exemplar_id": "awesome_evals::ae-081-promptfoo", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-081-promptfoo::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-081-promptfoo", "source_section": "5a · Eval frameworks & harnesses (code-first test-runners)", "source_title": "promptfoo", "source_type": "repository_or_docs", "source_url": "https://github.com/promptfoo/promptfoo"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://github.com/confident-ai/deepeval"], "content_sha256": "138fdabc91727fb2f32c3608a961d751c405e93ec1e2ccb56309c931c04a69d1", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-082-deepeval-confident-ai.raw.txt", "primary_url": "https://github.com/confident-ai/deepeval"}, "exemplar_id": "awesome_evals::ae-082-deepeval-confident-ai", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-082-deepeval-confident-ai::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-082-deepeval-confident-ai", "source_section": "5a · Eval frameworks & harnesses (code-first test-runners)", "source_title": "DeepEval / Confident AI", "source_type": "repository_or_docs", "source_url": "https://github.com/confident-ai/deepeval"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://github.com/pydantic/pydantic-ai"], "content_sha256": "4810e54183f6ab922d1dd31438ae8f1a6ba94950add254f667dca801d45c8503", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-083-pydantic-evals.raw.txt", "primary_url": "https://github.com/pydantic/pydantic-ai"}, "exemplar_id": "awesome_evals::ae-083-pydantic-evals", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-083-pydantic-evals::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-083-pydantic-evals", "source_section": "5a · Eval frameworks & harnesses (code-first test-runners)", "source_title": "pydantic-evals", "source_type": "repository_or_docs", "source_url": "https://github.com/pydantic/pydantic-ai"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://github.com/langchain-ai/openevals", "https://github.com/langchain-ai/agentevals"], "content_sha256": "60f08ff1afa489dc5208d482f87979268a4563b8cfb964587114455327d3d7d1", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-084-openevals.raw.txt", "primary_url": "https://github.com/langchain-ai/openevals"}, "exemplar_id": "awesome_evals::ae-084-openevals", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-084-openevals::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-084-openevals", "source_section": "5a · Eval frameworks & harnesses (code-first test-runners)", "source_title": "openevals", "source_type": "repository_or_docs", "source_url": "https://github.com/langchain-ai/openevals"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://mlflow.org/docs/latest/genai/eval-monitor/"], "content_sha256": "a28983ab44d522bae476d3a47e39d3728258e73a88ff5be96ce66d7f4b577d02", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-085-mlflow-genai-evaluate.raw.txt", "primary_url": "https://mlflow.org/docs/latest/genai/eval-monitor/"}, "exemplar_id": "awesome_evals::ae-085-mlflow-genai-evaluate", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-085-mlflow-genai-evaluate::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-085-mlflow-genai-evaluate", "source_section": "5a · Eval frameworks & harnesses (code-first test-runners)", "source_title": "MLflow GenAI evaluate", "source_type": "docs_or_book", "source_url": "https://mlflow.org/docs/latest/genai/eval-monitor/"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://github.com/stanford-crfm/helm"], "content_sha256": "c368539488eb44c518978d08f8e6ea31a78a7c5e3d3e06bc49bcbdae1ab96fb3", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-086-helm-crfm-helm.raw.txt", "primary_url": "https://github.com/stanford-crfm/helm"}, "exemplar_id": "awesome_evals::ae-086-helm-crfm-helm", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-086-helm-crfm-helm::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-086-helm-crfm-helm", "source_section": "5a · Eval frameworks & harnesses (code-first test-runners)", "source_title": "HELM (crfm-helm)", "source_type": "repository_or_docs", "source_url": "https://github.com/stanford-crfm/helm"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://github.com/Giskard-AI/giskard-oss"], "content_sha256": "bbee6cee85b50d463678f24200f11157d0098b951ad2d32ae452d00d952e1a36", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-087-giskard.raw.txt", "primary_url": "https://github.com/Giskard-AI/giskard-oss"}, "exemplar_id": "awesome_evals::ae-087-giskard", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-087-giskard::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-087-giskard", "source_section": "5a · Eval frameworks & harnesses (code-first test-runners)", "source_title": "Giskard", "source_type": "repository_or_docs", "source_url": "https://github.com/Giskard-AI/giskard-oss"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://github.com/deepchecks/deepchecks"], "content_sha256": "e2d2b4017f2687e2c13d89c5abb4ab494fd5694809a4be174e2895cd6a4aa226", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-088-deepchecks-llm.raw.txt", "primary_url": "https://github.com/deepchecks/deepchecks"}, "exemplar_id": "awesome_evals::ae-088-deepchecks-llm", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-088-deepchecks-llm::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-088-deepchecks-llm", "source_section": "5a · Eval frameworks & harnesses (code-first test-runners)", "source_title": "Deepchecks LLM", "source_type": "repository_or_docs", "source_url": "https://github.com/deepchecks/deepchecks"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://github.com/uptrain-ai/uptrain"], "content_sha256": "d645716799ccf96a17d75d6fe22262eec8fb6c7da87c6256272375670436d762", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-089-uptrain.raw.txt", "primary_url": "https://github.com/uptrain-ai/uptrain"}, "exemplar_id": "awesome_evals::ae-089-uptrain", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-089-uptrain::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-089-uptrain", "source_section": "5a · Eval frameworks & harnesses (code-first test-runners)", "source_title": "UpTrain", "source_type": "repository_or_docs", "source_url": "https://github.com/uptrain-ai/uptrain"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://github.com/huggingface/evaluate"], "content_sha256": "1698df1bc9babb362e22f66bde41743c967f8a9ef21b46bbf87d710aeec068a4", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-090-hf-evaluate.raw.txt", "primary_url": "https://github.com/huggingface/evaluate"}, "exemplar_id": "awesome_evals::ae-090-hf-evaluate", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-090-hf-evaluate::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-090-hf-evaluate", "source_section": "5a · Eval frameworks & harnesses (code-first test-runners)", "source_title": "HF `evaluate`", "source_type": "repository_or_docs", "source_url": "https://github.com/huggingface/evaluate"}
{"eval_domain": "harness_model_skill", "evidence": {"all_urls": ["https://github.com/harbor-framework/harbor"], "content_sha256": "8c60f53504cc837028ef3279a63b03f92a1167a7057749547820b6a4358c26a1", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-091-harbor.raw.txt", "primary_url": "https://github.com/harbor-framework/harbor"}, "exemplar_id": "awesome_evals::ae-091-harbor", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-091-harbor::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "harness_model_confusion", "source_id": "ae-091-harbor", "source_section": "5a · Eval frameworks & harnesses (code-first test-runners)", "source_title": "Harbor", "source_type": "repository_or_docs", "source_url": "https://github.com/harbor-framework/harbor"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://github.com/mattpocock/evalite"], "content_sha256": "7cf3a18126d185903d2ed7b6bf19810098f12acc30b7cf2b1602322e0ae22680", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-092-evalite.raw.txt", "primary_url": "https://github.com/mattpocock/evalite"}, "exemplar_id": "awesome_evals::ae-092-evalite", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-092-evalite::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-092-evalite", "source_section": "5b · TypeScript/JS-native eval runners", "source_title": "evalite", "source_type": "repository_or_docs", "source_url": "https://github.com/mattpocock/evalite"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://github.com/mastra-ai/mastra"], "content_sha256": "90be153179d780259348b1297d62f4a8faa501b05ecec1e2bc305878c73bb264", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/mastra-scorers.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-093-mastra-scorers.raw.txt", "primary_url": "https://github.com/mastra-ai/mastra"}, "exemplar_id": "awesome_evals::ae-093-mastra-scorers", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-093-mastra-scorers::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-093-mastra-scorers", "source_section": "5b · TypeScript/JS-native eval runners", "source_title": "Mastra scorers", "source_type": "repository_or_docs", "source_url": "https://github.com/mastra-ai/mastra"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://github.com/vercel-labs/agent-eval"], "content_sha256": "9f95f29b87ecb8e514d64bc9443ed8e727e89287c43be239d5d95fb32c20d7dc", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/vercel-eval-driven-development.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-094-vercel-agent-eval.raw.txt", "primary_url": "https://github.com/vercel-labs/agent-eval"}, "exemplar_id": "awesome_evals::ae-094-vercel-agent-eval", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-094-vercel-agent-eval::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-094-vercel-agent-eval", "source_section": "5b · TypeScript/JS-native eval runners", "source_title": "Vercel agent-eval", "source_type": "repository_or_docs", "source_url": "https://github.com/vercel-labs/agent-eval"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://github.com/braintrustdata/autoevals"], "content_sha256": "e73aaacece7f23ee8777537c5c6860131566aee13da1adc77eefa71ce7b1ceec", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-095-autoevals.raw.txt", "primary_url": "https://github.com/braintrustdata/autoevals"}, "exemplar_id": "awesome_evals::ae-095-autoevals", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-095-autoevals::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-095-autoevals", "source_section": "5b · TypeScript/JS-native eval runners", "source_title": "Autoevals", "source_type": "repository_or_docs", "source_url": "https://github.com/braintrustdata/autoevals"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://github.com/truera/trulens"], "content_sha256": "49272ff8a86e1bb5ff189ce5c09ad5c546e63880913647ff0fada21e16a39e9b", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-096-trulens.raw.txt", "primary_url": "https://github.com/truera/trulens"}, "exemplar_id": "awesome_evals::ae-096-trulens", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-096-trulens::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-096-trulens", "source_section": "5c · RAG / retrieval evaluation", "source_title": "TruLens", "source_type": "repository_or_docs", "source_url": "https://github.com/truera/trulens"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://github.com/stanford-futuredata/ARES"], "content_sha256": "5ecf26044ff0caa518136549b4762fda48246c1fe216c78a9793b3d05cc07302", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-097-ares.raw.txt", "primary_url": "https://github.com/stanford-futuredata/ARES"}, "exemplar_id": "awesome_evals::ae-097-ares", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-097-ares::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-097-ares", "source_section": "5c · RAG / retrieval evaluation", "source_title": "ARES", "source_type": "repository_or_docs", "source_url": "https://github.com/stanford-futuredata/ARES"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://github.com/amazon-science/RAGChecker"], "content_sha256": "9d087e101857395f7c5d0b2a93d49e8a9f765282b2f2db2931813f5dd5809e15", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-098-ragchecker.raw.txt", "primary_url": "https://github.com/amazon-science/RAGChecker"}, "exemplar_id": "awesome_evals::ae-098-ragchecker", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-098-ragchecker::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-098-ragchecker", "source_section": "5c · RAG / retrieval evaluation", "source_title": "RAGChecker", "source_type": "repository_or_docs", "source_url": "https://github.com/amazon-science/RAGChecker"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://github.com/relari-ai/continuous-eval"], "content_sha256": "2e97ee3b6ef828fbf47b20537d8dc2593c7045852ba6d668ea6ea5ee3dd18d17", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-099-continuous-eval-relari.raw.txt", "primary_url": "https://github.com/relari-ai/continuous-eval"}, "exemplar_id": "awesome_evals::ae-099-continuous-eval-relari", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-099-continuous-eval-relari::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-099-continuous-eval-relari", "source_section": "5c · RAG / retrieval evaluation", "source_title": "continuous-eval (Relari)", "source_type": "repository_or_docs", "source_url": "https://github.com/relari-ai/continuous-eval"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://github.com/TonicAI/tonic_validate"], "content_sha256": "308c4c6bbeb8d3cb7aa99fec9a53706d84f1a15f7032f8fc52148726df90b087", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-100-tonic-validate.raw.txt", "primary_url": "https://github.com/TonicAI/tonic_validate"}, "exemplar_id": "awesome_evals::ae-100-tonic-validate", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-100-tonic-validate::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-100-tonic-validate", "source_section": "5c · RAG / retrieval evaluation", "source_title": "Tonic Validate", "source_type": "repository_or_docs", "source_url": "https://github.com/TonicAI/tonic_validate"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://github.com/haizelabs/verdict"], "content_sha256": "5897c65e973b2c5aff5c7c17a42fdecfffcddcd284d5c85511b1b13dbe76717c", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-101-verdict.raw.txt", "primary_url": "https://github.com/haizelabs/verdict"}, "exemplar_id": "awesome_evals::ae-101-verdict", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-101-verdict::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-101-verdict", "source_section": "5d · LLM-as-judge / reward / verifier libraries", "source_title": "verdict", "source_type": "repository_or_docs", "source_url": "https://github.com/haizelabs/verdict"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://github.com/OpenPipe/ART"], "content_sha256": "1ef910fb21451ba797c0abfa41efe10aee4ebec99e26bef93e78c07e29614167", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-102-ruler.raw.txt", "primary_url": "https://github.com/OpenPipe/ART"}, "exemplar_id": "awesome_evals::ae-102-ruler", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-102-ruler::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-102-ruler", "source_section": "5d · LLM-as-judge / reward / verifier libraries", "source_title": "RULER", "source_type": "repository_or_docs", "source_url": "https://github.com/OpenPipe/ART"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://github.com/prometheus-eval/prometheus-eval"], "content_sha256": "6619fe0ac8084236d29de42dd627a42e52d923c3b489bf8bb95b6245cbf69e0c", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-103-prometheus-2.raw.txt", "primary_url": "https://github.com/prometheus-eval/prometheus-eval"}, "exemplar_id": "awesome_evals::ae-103-prometheus-2", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-103-prometheus-2::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-103-prometheus-2", "source_section": "5d · LLM-as-judge / reward / verifier libraries", "source_title": "Prometheus 2", "source_type": "repository_or_docs", "source_url": "https://github.com/prometheus-eval/prometheus-eval"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://github.com/atla-ai/selene-mini"], "content_sha256": "4ae6e58dda6d11aa0701591bb31d363c7e16bcfad4513308ccc1f1f9fdae4d98", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-104-atla-selene.raw.txt", "primary_url": "https://github.com/atla-ai/selene-mini"}, "exemplar_id": "awesome_evals::ae-104-atla-selene", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-104-atla-selene::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-104-atla-selene", "source_section": "5d · LLM-as-judge / reward / verifier libraries", "source_title": "Atla Selene", "source_type": "repository_or_docs", "source_url": "https://github.com/atla-ai/selene-mini"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://github.com/patronus-ai/Lynx-hallucination-detection", "https://github.com/patronus-ai/glider"], "content_sha256": "ed2fadd4540805786da2ba2ec0ca6c5e621597b0a7318afe243d5d286d0a4dcb", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-105-patronus-lynx-glider.raw.txt", "primary_url": "https://github.com/patronus-ai/Lynx-hallucination-detection"}, "exemplar_id": "awesome_evals::ae-105-patronus-lynx-glider", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-105-patronus-lynx-glider::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-105-patronus-lynx-glider", "source_section": "5d · LLM-as-judge / reward / verifier libraries", "source_title": "Patronus Lynx / GLIDER", "source_type": "repository_or_docs", "source_url": "https://github.com/patronus-ai/Lynx-hallucination-detection"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://github.com/flowaicom/flow-judge"], "content_sha256": "61f709c29407b9b021e2cdb3b09f571441974955b8ab7813e1d2042436cb6c43", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-106-flow-judge.raw.txt", "primary_url": "https://github.com/flowaicom/flow-judge"}, "exemplar_id": "awesome_evals::ae-106-flow-judge", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-106-flow-judge::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-106-flow-judge", "source_section": "5d · LLM-as-judge / reward / verifier libraries", "source_title": "Flow-Judge", "source_type": "repository_or_docs", "source_url": "https://github.com/flowaicom/flow-judge"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://github.com/allenai/reward-bench"], "content_sha256": "6d0614ea70ae1d34fd5d274a5f37e03fe113967d16188dcf1fba94451739e3ca", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-107-rewardbench.raw.txt", "primary_url": "https://github.com/allenai/reward-bench"}, "exemplar_id": "awesome_evals::ae-107-rewardbench", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-107-rewardbench::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-107-rewardbench", "source_section": "5d · LLM-as-judge / reward / verifier libraries", "source_title": "RewardBench", "source_type": "repository_or_docs", "source_url": "https://github.com/allenai/reward-bench"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://github.com/ScalerLab/JudgeBench"], "content_sha256": "c8dde52ad92b885b53fbe7a2210768289008ace4f8adbcdd6dc735dfb6dd8d90", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-108-judgebench.raw.txt", "primary_url": "https://github.com/ScalerLab/JudgeBench"}, "exemplar_id": "awesome_evals::ae-108-judgebench", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-108-judgebench::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-108-judgebench", "source_section": "5d · LLM-as-judge / reward / verifier libraries", "source_title": "JudgeBench", "source_type": "repository_or_docs", "source_url": "https://github.com/ScalerLab/JudgeBench"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://github.com/fw-ai-external/reward-kit"], "content_sha256": "eef535822a6567ef0129d1624a1f79e758e734fed13e8b7c9bfb505d07532098", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-109-reward-kit.raw.txt", "primary_url": "https://github.com/fw-ai-external/reward-kit"}, "exemplar_id": "awesome_evals::ae-109-reward-kit", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-109-reward-kit::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-109-reward-kit", "source_section": "5d · LLM-as-judge / reward / verifier libraries", "source_title": "reward-kit", "source_type": "repository_or_docs", "source_url": "https://github.com/fw-ai-external/reward-kit"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://github.com/PrimeIntellect-ai/verifiers"], "content_sha256": "d09eef5e271a97ac198fbaaab075ce75bab24113f95892b1dfd744da54dbff41", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-110-verifiers.raw.txt", "primary_url": "https://github.com/PrimeIntellect-ai/verifiers"}, "exemplar_id": "awesome_evals::ae-110-verifiers", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-110-verifiers::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-110-verifiers", "source_section": "5e · RL-environment / verifiable-reward toolkits (eval ⇄ training)", "source_title": "verifiers", "source_type": "repository_or_docs", "source_url": "https://github.com/PrimeIntellect-ai/verifiers"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://github.com/PrimeIntellect-ai/community-environments"], "content_sha256": "c452122e496646c18fbeddfbe45405b34ad2149361fb32f5b14bc737f19711d2", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-111-environments-hub.raw.txt", "primary_url": "https://github.com/PrimeIntellect-ai/community-environments"}, "exemplar_id": "awesome_evals::ae-111-environments-hub", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-111-environments-hub::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-111-environments-hub", "source_section": "5e · RL-environment / verifiable-reward toolkits (eval ⇄ training)", "source_title": "Environments Hub", "source_type": "repository_or_docs", "source_url": "https://github.com/PrimeIntellect-ai/community-environments"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://github.com/PrimeIntellect-ai/prime-rl"], "content_sha256": "271ab5786e015572ea5f684385667d3db081c0fd045ab7b65bc8b46d821717d3", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-112-prime-rl.raw.txt", "primary_url": "https://github.com/PrimeIntellect-ai/prime-rl"}, "exemplar_id": "awesome_evals::ae-112-prime-rl", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-112-prime-rl::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-112-prime-rl", "source_section": "5e · RL-environment / verifiable-reward toolkits (eval ⇄ training)", "source_title": "prime-rl", "source_type": "repository_or_docs", "source_url": "https://github.com/PrimeIntellect-ai/prime-rl"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://github.com/benchflow-ai/benchflow", "https://benchflow.ai"], "content_sha256": "a8d7201002d55117130f3af7b0d978adc15e76798501a2ecf8d8b7c1893ffd84", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-113-benchflow.raw.txt", "primary_url": "https://github.com/benchflow-ai/benchflow"}, "exemplar_id": "awesome_evals::ae-113-benchflow", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-113-benchflow::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-113-benchflow", "source_section": "5e · RL-environment / verifiable-reward toolkits (eval ⇄ training)", "source_title": "BenchFlow", "source_type": "repository_or_docs", "source_url": "https://github.com/benchflow-ai/benchflow"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://github.com/hud-evals/hud-python"], "content_sha256": "29e0838394a9059ad981edeb8cb5595e83cf40ca934b1a2a75052052bc129f35", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-114-hud.raw.txt", "primary_url": "https://github.com/hud-evals/hud-python"}, "exemplar_id": "awesome_evals::ae-114-hud", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-114-hud::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-114-hud", "source_section": "5e · RL-environment / verifiable-reward toolkits (eval ⇄ training)", "source_title": "HUD", "source_type": "repository_or_docs", "source_url": "https://github.com/hud-evals/hud-python"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://github.com/NousResearch/atropos"], "content_sha256": "5016664f26ece61a80271e10b03d01a974641343d71dc83197bebe15c71c4c74", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-115-atropos.raw.txt", "primary_url": "https://github.com/NousResearch/atropos"}, "exemplar_id": "awesome_evals::ae-115-atropos", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-115-atropos::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-115-atropos", "source_section": "5e · RL-environment / verifiable-reward toolkits (eval ⇄ training)", "source_title": "Atropos", "source_type": "repository_or_docs", "source_url": "https://github.com/NousResearch/atropos"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://github.com/volcengine/verl"], "content_sha256": "632c2103709f8b5c0e080c54e039cc232f337d3e6541c2afb82943943496b77a", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-116-verl.raw.txt", "primary_url": "https://github.com/volcengine/verl"}, "exemplar_id": "awesome_evals::ae-116-verl", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-116-verl::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-116-verl", "source_section": "5e · RL-environment / verifiable-reward toolkits (eval ⇄ training)", "source_title": "verl", "source_type": "repository_or_docs", "source_url": "https://github.com/volcengine/verl"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://github.com/OpenRLHF/OpenRLHF", "https://github.com/NovaSky-AI/SkyRL", "https://github.com/areal-project/AReaL", "https://github.com/alibaba/ROLL", "https://github.com/agentica-project/rllm", "https://github.com/huggingface/trl"], "content_sha256": "e971ac54c8f6ddfde36af3927e55a1022214deed51c79c0f1725f577ead52b4e", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-117-openrlhf.raw.txt", "primary_url": "https://github.com/OpenRLHF/OpenRLHF"}, "exemplar_id": "awesome_evals::ae-117-openrlhf", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-117-openrlhf::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-117-openrlhf", "source_section": "5e · RL-environment / verifiable-reward toolkits (eval ⇄ training)", "source_title": "OpenRLHF", "source_type": "repository_or_docs", "source_url": "https://github.com/OpenRLHF/OpenRLHF"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://docs.openreward.ai/"], "content_sha256": "464119c1a768a7b4a908c726262f01de7cea66c687e282c74f85246984126daf", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/open-reward-standard-ors.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-118-open-reward-standard-ors.raw.txt", "primary_url": "https://docs.openreward.ai/"}, "exemplar_id": "awesome_evals::ae-118-open-reward-standard-ors", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-118-open-reward-standard-ors::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-118-open-reward-standard-ors", "source_section": "5e · RL-environment / verifiable-reward toolkits (eval ⇄ training)", "source_title": "Open Reward Standard (ORS)", "source_type": "docs_or_book", "source_url": "https://docs.openreward.ai/"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://github.com/Arize-ai/phoenix"], "content_sha256": "17a97d209b10488f437cc8dcdd421fef271bb7c9854ca560a280bd484f010e2d", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-119-arize-phoenix.raw.txt", "primary_url": "https://github.com/Arize-ai/phoenix"}, "exemplar_id": "awesome_evals::ae-119-arize-phoenix", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-119-arize-phoenix::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-119-arize-phoenix", "source_section": "5f · Observability + eval platforms (tracing · datasets · online/offline · CI)", "source_title": "Arize Phoenix", "source_type": "repository_or_docs", "source_url": "https://github.com/Arize-ai/phoenix"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://github.com/langfuse/langfuse"], "content_sha256": "2170f043b4e46002ed67703080fe2586db9c5e76400b89350ea815720a8de7d5", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-120-langfuse.raw.txt", "primary_url": "https://github.com/langfuse/langfuse"}, "exemplar_id": "awesome_evals::ae-120-langfuse", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-120-langfuse::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-120-langfuse", "source_section": "5f · Observability + eval platforms (tracing · datasets · online/offline · CI)", "source_title": "Langfuse", "source_type": "repository_or_docs", "source_url": "https://github.com/langfuse/langfuse"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://github.com/comet-ml/opik"], "content_sha256": "48682867dc0c36d8e5efead765b9779b824c6b32b4aad2ab40df1600eebbc25f", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-121-opik.raw.txt", "primary_url": "https://github.com/comet-ml/opik"}, "exemplar_id": "awesome_evals::ae-121-opik", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-121-opik::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-121-opik", "source_section": "5f · Observability + eval platforms (tracing · datasets · online/offline · CI)", "source_title": "Opik", "source_type": "repository_or_docs", "source_url": "https://github.com/comet-ml/opik"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://github.com/wandb/weave"], "content_sha256": "b80f51e794763add6a6ff7285e933e8dbcfe911b75a5cb3b40d23c3686698d70", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-122-w-b-weave.raw.txt", "primary_url": "https://github.com/wandb/weave"}, "exemplar_id": "awesome_evals::ae-122-w-b-weave", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-122-w-b-weave::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-122-w-b-weave", "source_section": "5f · Observability + eval platforms (tracing · datasets · online/offline · CI)", "source_title": "W&B Weave", "source_type": "repository_or_docs", "source_url": "https://github.com/wandb/weave"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://www.braintrust.dev/docs/start/eval-sdk"], "content_sha256": "8c8a88c65d922b19f100e94c8a5081dc21a16abbaeb049896a76f3c60c3d8eda", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-123-braintrust.raw.txt", "primary_url": "https://www.braintrust.dev/docs/start/eval-sdk"}, "exemplar_id": "awesome_evals::ae-123-braintrust", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-123-braintrust::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-123-braintrust", "source_section": "5f · Observability + eval platforms (tracing · datasets · online/offline · CI)", "source_title": "Braintrust", "source_type": "docs_or_book", "source_url": "https://www.braintrust.dev/docs/start/eval-sdk"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://www.patronus.ai/"], "content_sha256": "e5a0ff2ed2a9f15ea9db2376f5e748400151c3df960d12968868f41e81605680", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-124-patronus-ai.raw.txt", "primary_url": "https://www.patronus.ai/"}, "exemplar_id": "awesome_evals::ae-124-patronus-ai", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-124-patronus-ai::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-124-patronus-ai", "source_section": "5f · Observability + eval platforms (tracing · datasets · online/offline · CI)", "source_title": "Patronus AI", "source_type": "web_article", "source_url": "https://www.patronus.ai/"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://www.getmaxim.ai/"], "content_sha256": "a267fa4b50260a93385c06e03ee31890b35a9b74aebf98e6435e93d4becace4e", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-125-maxim-ai.raw.txt", "primary_url": "https://www.getmaxim.ai/"}, "exemplar_id": "awesome_evals::ae-125-maxim-ai", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-125-maxim-ai::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-125-maxim-ai", "source_section": "5f · Observability + eval platforms (tracing · datasets · online/offline · CI)", "source_title": "Maxim AI", "source_type": "web_article", "source_url": "https://www.getmaxim.ai/"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://galileo.ai/"], "content_sha256": "ccacb82db80d71a8bfaf920007db1112e53a3ec05b51bcb1fc8e54ffb8a88d7f", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-126-galileo.raw.txt", "primary_url": "https://galileo.ai/"}, "exemplar_id": "awesome_evals::ae-126-galileo", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-126-galileo::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-126-galileo", "source_section": "5f · Observability + eval platforms (tracing · datasets · online/offline · CI)", "source_title": "Galileo", "source_type": "web_article", "source_url": "https://galileo.ai/"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://www.vellum.ai/"], "content_sha256": "b33d7dce928a99b47b6f17dad0f78d4f25f452665e61e9f9c361f20bee4f82b5", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-127-vellum.raw.txt", "primary_url": "https://www.vellum.ai/"}, "exemplar_id": "awesome_evals::ae-127-vellum", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-127-vellum::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-127-vellum", "source_section": "5f · Observability + eval platforms (tracing · datasets · online/offline · CI)", "source_title": "Vellum", "source_type": "web_article", "source_url": "https://www.vellum.ai/"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://github.com/helicone/helicone"], "content_sha256": "165412284d01732a0a8a9bf8b0600cc8eefac05f14799a0d61402b52be2936c3", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-128-helicone.raw.txt", "primary_url": "https://github.com/helicone/helicone"}, "exemplar_id": "awesome_evals::ae-128-helicone", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-128-helicone::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-128-helicone", "source_section": "5f · Observability + eval platforms (tracing · datasets · online/offline · CI)", "source_title": "Helicone", "source_type": "repository_or_docs", "source_url": "https://github.com/helicone/helicone"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://github.com/traceloop/openllmetry"], "content_sha256": "38d6eeda187cf5793ba53fad0c9daa7cd45d2445380c4cedcf0decb0e2a76458", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-129-traceloop-openllmetry.raw.txt", "primary_url": "https://github.com/traceloop/openllmetry"}, "exemplar_id": "awesome_evals::ae-129-traceloop-openllmetry", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-129-traceloop-openllmetry::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-129-traceloop-openllmetry", "source_section": "5f · Observability + eval platforms (tracing · datasets · online/offline · CI)", "source_title": "Traceloop / OpenLLMetry", "source_type": "repository_or_docs", "source_url": "https://github.com/traceloop/openllmetry"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://github.com/Scale3-Labs/langtrace"], "content_sha256": "cbca8691cdccf80ae1a0c1adc962978a6021a2604f86f7b0e7cbaf477e2510de", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-130-langtrace.raw.txt", "primary_url": "https://github.com/Scale3-Labs/langtrace"}, "exemplar_id": "awesome_evals::ae-130-langtrace", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-130-langtrace::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-130-langtrace", "source_section": "5f · Observability + eval platforms (tracing · datasets · online/offline · CI)", "source_title": "Langtrace", "source_type": "repository_or_docs", "source_url": "https://github.com/Scale3-Labs/langtrace"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://github.com/whylabs/langkit"], "content_sha256": "ab6f2bc9cee3bafe9d295433500ba87de1c78571c3c35d0499b241e48635ca90", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-131-whylabs-langkit.raw.txt", "primary_url": "https://github.com/whylabs/langkit"}, "exemplar_id": "awesome_evals::ae-131-whylabs-langkit", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-131-whylabs-langkit::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-131-whylabs-langkit", "source_section": "5f · Observability + eval platforms (tracing · datasets · online/offline · CI)", "source_title": "WhyLabs / LangKit", "source_type": "repository_or_docs", "source_url": "https://github.com/whylabs/langkit"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://github.com/portkey-ai/gateway"], "content_sha256": "7543749e8d864eedfdca1b7ae762373ebafa30bf775bbe6e0a9072eab21a7c68", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-132-portkey.raw.txt", "primary_url": "https://github.com/portkey-ai/gateway"}, "exemplar_id": "awesome_evals::ae-132-portkey", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-132-portkey::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-132-portkey", "source_section": "5f · Observability + eval platforms (tracing · datasets · online/offline · CI)", "source_title": "Portkey", "source_type": "repository_or_docs", "source_url": "https://github.com/portkey-ai/gateway"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://www.datadoghq.com/product/ai/llm-observability/"], "content_sha256": "11740908ac0dc1fc09b3ed9b2e39581b9a12f049e43c61a02c1c6db1256b39a2", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-133-datadog-llm-observability.raw.txt", "primary_url": "https://www.datadoghq.com/product/ai/llm-observability/"}, "exemplar_id": "awesome_evals::ae-133-datadog-llm-observability", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-133-datadog-llm-observability::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-133-datadog-llm-observability", "source_section": "5f · Observability + eval platforms (tracing · datasets · online/offline · CI)", "source_title": "Datadog LLM Observability", "source_type": "web_article", "source_url": "https://www.datadoghq.com/product/ai/llm-observability/"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://www.fiddler.ai/"], "content_sha256": "6c3cbe8bc5c9464cae706696221bd3cd876fe356e817202b484d02400e61e05e", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-134-fiddler-ai.raw.txt", "primary_url": "https://www.fiddler.ai/"}, "exemplar_id": "awesome_evals::ae-134-fiddler-ai", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-134-fiddler-ai::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-134-fiddler-ai", "source_section": "5f · Observability + eval platforms (tracing · datasets · online/offline · CI)", "source_title": "Fiddler AI", "source_type": "web_article", "source_url": "https://www.fiddler.ai/"}
{"eval_domain": "observability_surface", "evidence": {"all_urls": ["https://www.promptlayer.com/", "https://newrelic.com/platform/ai-monitoring"], "content_sha256": "bc0a2d71bb8b04d188f19c24bf606d24bd2f693de945674502130fce18224d0e", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-135-promptlayer.raw.txt", "primary_url": "https://www.promptlayer.com/"}, "exemplar_id": "awesome_evals::ae-135-promptlayer", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-135-promptlayer::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "missing_trace", "source_id": "ae-135-promptlayer", "source_section": "5f · Observability + eval platforms (tracing · datasets · online/offline · CI)", "source_title": "PromptLayer", "source_type": "web_article", "source_url": "https://www.promptlayer.com/"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://github.com/Arize-ai/openinference"], "content_sha256": "44cf49a50e2443f7d65631deaadf56a1a8efb553a49d516574de5c3c02d7f2db", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-136-openinference.raw.txt", "primary_url": "https://github.com/Arize-ai/openinference"}, "exemplar_id": "awesome_evals::ae-136-openinference", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-136-openinference::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-136-openinference", "source_section": "5g · Tracing standards", "source_title": "OpenInference", "source_type": "repository_or_docs", "source_url": "https://github.com/Arize-ai/openinference"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://opentelemetry.io/docs/specs/semconv/gen-ai/"], "content_sha256": "f8f5ac9a7e5a7bbbd7a836c5d3b2c1d457ebecf9de9e05e57793bd64d17d6848", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-137-opentelemetry-genai-semantic-conventions.raw.txt", "primary_url": "https://opentelemetry.io/docs/specs/semconv/gen-ai/"}, "exemplar_id": "awesome_evals::ae-137-opentelemetry-genai-semantic-conventions", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-137-opentelemetry-genai-semantic-conventions::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-137-opentelemetry-genai-semantic-conventions", "source_section": "5g · Tracing standards", "source_title": "OpenTelemetry GenAI semantic conventions", "source_type": "docs_or_book", "source_url": "https://opentelemetry.io/docs/specs/semconv/gen-ai/"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://www.braintrust.dev/"], "content_sha256": "b076e975e3d0cf170adfbaf28313351be5ed023efc4090a179835d260f8a243a", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-138-braintrust.raw.txt", "primary_url": "https://www.braintrust.dev/"}, "exemplar_id": "awesome_evals::ae-138-braintrust", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-138-braintrust::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-138-braintrust", "source_section": "5g · Tracing standards", "source_title": "Braintrust", "source_type": "web_article", "source_url": "https://www.braintrust.dev/"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://github.com/raga-ai-hub/RagaAI-Catalyst"], "content_sha256": "dac44c12d5efb68b9ff150c799552333e06dba38cccaafa81c42988db8953d11", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-139-ragaai-catalyst.raw.txt", "primary_url": "https://github.com/raga-ai-hub/RagaAI-Catalyst"}, "exemplar_id": "awesome_evals::ae-139-ragaai-catalyst", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-139-ragaai-catalyst::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-139-ragaai-catalyst", "source_section": "5g · Tracing standards", "source_title": "RagaAI Catalyst", "source_type": "repository_or_docs", "source_url": "https://github.com/raga-ai-hub/RagaAI-Catalyst"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://developers.openai.com/cookbook/topic/evals"], "content_sha256": "99b818cd6e3ab4648a01ee8b44f4a7bf3bb2d70f63276649010c57eeb4f93370", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/macro-evals-agentic-systems-openai-cookbook.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-140-openai-cookbook-evals.raw.txt", "primary_url": "https://developers.openai.com/cookbook/topic/evals"}, "exemplar_id": "awesome_evals::ae-140-openai-cookbook-evals", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-140-openai-cookbook-evals::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-140-openai-cookbook-evals", "source_section": "5g · Tracing standards", "source_title": "OpenAI Cookbook — Evals", "source_type": "docs_or_book", "source_url": "https://developers.openai.com/cookbook/topic/evals"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://ofir.io/How-to-Build-Good-Language-Modeling-Benchmarks/"], "content_sha256": "cb4d4263c57f9558283b48d61549aa99a8931294452efa8cbe2860c1b4c0b7cd", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-141-how-to-build-good-language-modeling-benchmarks.raw.txt", "primary_url": "https://ofir.io/How-to-Build-Good-Language-Modeling-Benchmarks/"}, "exemplar_id": "awesome_evals::ae-141-how-to-build-good-language-modeling-benchmarks", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-141-how-to-build-good-language-modeling-benchmarks::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-141-how-to-build-good-language-modeling-benchmarks", "source_section": "6 · Benchmark vs. eval (and benchmark integrity: contamination, saturation, label errors, leaderboard gaming)", "source_title": "How to Build Good Language Modeling Benchmarks", "source_type": "web_article", "source_url": "https://ofir.io/How-to-Build-Good-Language-Modeling-Benchmarks/"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://arxiv.org/abs/2407.01502"], "content_sha256": "a74a66bddeea47fd5b656b5bd316858bb9baf7b951491316ec773e834c43f53c", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-142-ai-agents-that-matter.raw.txt", "primary_url": "https://arxiv.org/abs/2407.01502"}, "exemplar_id": "awesome_evals::ae-142-ai-agents-that-matter", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-142-ai-agents-that-matter::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-142-ai-agents-that-matter", "source_section": "6 · Benchmark vs. eval (and benchmark integrity: contamination, saturation, label errors, leaderboard gaming)", "source_title": "AI Agents That Matter", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2407.01502"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://openai.com/index/why-we-no-longer-evaluate-swe-bench-verified/", "https://decrypt.co/359012/"], "content_sha256": "0784dd5613c68bdc2a53ab8c6abd0e98834c52c2544df4224144c4d0ee0b5d8b", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-143-why-we-no-longer-evaluate-swe-bench-verified.raw.txt", "primary_url": "https://openai.com/index/why-we-no-longer-evaluate-swe-bench-verified/"}, "exemplar_id": "awesome_evals::ae-143-why-we-no-longer-evaluate-swe-bench-verified", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-143-why-we-no-longer-evaluate-swe-bench-verified::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-143-why-we-no-longer-evaluate-swe-bench-verified", "source_section": "6 · Benchmark vs. eval (and benchmark integrity: contamination, saturation, label errors, leaderboard gaming)", "source_title": "Why We No Longer Evaluate SWE-bench Verified", "source_type": "web_article", "source_url": "https://openai.com/index/why-we-no-longer-evaluate-swe-bench-verified/"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://arxiv.org/abs/2504.20879"], "content_sha256": "3d80cddbaf82f9841da2961f528d89ed22fc95db154019ef6359b6cc6601cfd7", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/leaderboard-illusion.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-144-the-leaderboard-illusion.raw.txt", "primary_url": "https://arxiv.org/abs/2504.20879"}, "exemplar_id": "awesome_evals::ae-144-the-leaderboard-illusion", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-144-the-leaderboard-illusion::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-144-the-leaderboard-illusion", "source_section": "6 · Benchmark vs. eval (and benchmark integrity: contamination, saturation, label errors, leaderboard gaming)", "source_title": "The Leaderboard Illusion", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2504.20879"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://arxiv.org/abs/2506.12286"], "content_sha256": "085ac6c81b4d33668e3405b95a76e26c79a3925531c3a17fc16abf232df06a20", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-145-the-swe-bench-illusion-when-sota-llms-remember-i.raw.txt", "primary_url": "https://arxiv.org/abs/2506.12286"}, "exemplar_id": "awesome_evals::ae-145-the-swe-bench-illusion-when-sota-llms-remember-i", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-145-the-swe-bench-illusion-when-sota-llms-remember-i::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-145-the-swe-bench-illusion-when-sota-llms-remember-i", "source_section": "6 · Benchmark vs. eval (and benchmark integrity: contamination, saturation, label errors, leaderboard gaming)", "source_title": "The SWE-bench Illusion: When SOTA LLMs Remember Instead of Reason", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2506.12286"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://arxiv.org/abs/2507.02825"], "content_sha256": "ea24245370ed1f98005c7bf00b44411763d9488744f9a586ebee821dd8c59236", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-146-establishing-best-practices-for-building-rigorou.raw.txt", "primary_url": "https://arxiv.org/abs/2507.02825"}, "exemplar_id": "awesome_evals::ae-146-establishing-best-practices-for-building-rigorou", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-146-establishing-best-practices-for-building-rigorou::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-146-establishing-best-practices-for-building-rigorou", "source_section": "6 · Benchmark vs. eval (and benchmark integrity: contamination, saturation, label errors, leaderboard gaming)", "source_title": "Establishing Best Practices for Building Rigorous Agentic Benchmarks (ABC)", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2507.02825"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://epoch.ai/benchmarks/frontiermath-tiers-1-3-v2"], "content_sha256": "9ba719b1cdda1396b760d66d0ac929dff907ee42f5e9b60a3129238418f08509", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-147-frontiermath-tiers-1-3-v2-corrected.raw.txt", "primary_url": "https://epoch.ai/benchmarks/frontiermath-tiers-1-3-v2"}, "exemplar_id": "awesome_evals::ae-147-frontiermath-tiers-1-3-v2-corrected", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-147-frontiermath-tiers-1-3-v2-corrected::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-147-frontiermath-tiers-1-3-v2-corrected", "source_section": "6 · Benchmark vs. eval (and benchmark integrity: contamination, saturation, label errors, leaderboard gaming)", "source_title": "FrontierMath Tiers 1–3 v2 (corrected)", "source_type": "web_article", "source_url": "https://epoch.ai/benchmarks/frontiermath-tiers-1-3-v2"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://www.futurehouse.org/research-announcements/hle-exam", "https://www.lesswrong.com/posts/JANqfGrMyBgcKtGgK/"], "content_sha256": "c1e184a78842724b34abe6c747a3dd88e12303edb0529fb06e0365e0b55dc5a6", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-148-about-30-of-humanity-s-last-exam-answers-are-wro.raw.txt", "primary_url": "https://www.futurehouse.org/research-announcements/hle-exam"}, "exemplar_id": "awesome_evals::ae-148-about-30-of-humanity-s-last-exam-answers-are-wro", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-148-about-30-of-humanity-s-last-exam-answers-are-wro::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-148-about-30-of-humanity-s-last-exam-answers-are-wro", "source_section": "6 · Benchmark vs. eval (and benchmark integrity: contamination, saturation, label errors, leaderboard gaming)", "source_title": "About 30% of Humanity's Last Exam Answers Are Wrong", "source_type": "web_article", "source_url": "https://www.futurehouse.org/research-announcements/hle-exam"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://www.interconnects.ai/p/building-on-evaluation-quicksand"], "content_sha256": "b9885f095cf3cc54596b20530d30dbfae608ada2f47dec0aaa19cd751e883fb6", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-149-building-on-evaluation-quicksand.raw.txt", "primary_url": "https://www.interconnects.ai/p/building-on-evaluation-quicksand"}, "exemplar_id": "awesome_evals::ae-149-building-on-evaluation-quicksand", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-149-building-on-evaluation-quicksand::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-149-building-on-evaluation-quicksand", "source_section": "6 · Benchmark vs. eval (and benchmark integrity: contamination, saturation, label errors, leaderboard gaming)", "source_title": "Building on Evaluation Quicksand", "source_type": "blog", "source_url": "https://www.interconnects.ai/p/building-on-evaluation-quicksand"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://arxiv.org/abs/2601.17087"], "content_sha256": "35e70774dda9cebce0a3ab5ca52b9fcf77475c3499ee58a6e79083dce88a5fab", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-150-lost-in-simulation.raw.txt", "primary_url": "https://arxiv.org/abs/2601.17087"}, "exemplar_id": "awesome_evals::ae-150-lost-in-simulation", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-150-lost-in-simulation::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-150-lost-in-simulation", "source_section": "6 · Benchmark vs. eval (and benchmark integrity: contamination, saturation, label errors, leaderboard gaming)", "source_title": "Lost in Simulation", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2601.17087"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://arxiv.org/abs/2310.06770", "https://www.swebench.com"], "content_sha256": "998a92461b56f5d6109a4c9161c6a2a12649f95ca14715b0a439bfcf01f16885", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-151-swe-bench-can-lms-resolve-real-world-github-issu.raw.txt", "primary_url": "https://arxiv.org/abs/2310.06770"}, "exemplar_id": "awesome_evals::ae-151-swe-bench-can-lms-resolve-real-world-github-issu", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-151-swe-bench-can-lms-resolve-real-world-github-issu::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-151-swe-bench-can-lms-resolve-real-world-github-issu", "source_section": "6 · Benchmark vs. eval (and benchmark integrity: contamination, saturation, label errors, leaderboard gaming)", "source_title": "SWE-bench: Can LMs Resolve Real-World GitHub Issues?", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2310.06770"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://eugeneyan.com/writing/evals/"], "content_sha256": "8d24abf3315225ae2d2cba21ec7aca650bf76751a34ff5756e066f87b97662fb", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-karam-metrics-that-work.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-152-task-specific-llm-evals-that-do-don-t-work.raw.txt", "primary_url": "https://eugeneyan.com/writing/evals/"}, "exemplar_id": "awesome_evals::ae-152-task-specific-llm-evals-that-do-don-t-work", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-152-task-specific-llm-evals-that-do-don-t-work::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-152-task-specific-llm-evals-that-do-don-t-work", "source_section": "6 · Benchmark vs. eval (and benchmark integrity: contamination, saturation, label errors, leaderboard gaming)", "source_title": "Task-Specific LLM Evals that Do & Don't Work", "source_type": "web_article", "source_url": "https://eugeneyan.com/writing/evals/"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://x.com/karpathy/status/1896266683301659068"], "content_sha256": null, "http_status": null, "local_note_path": null, "local_raw_path": null, "primary_url": "https://x.com/karpathy/status/1896266683301659068"}, "exemplar_id": "awesome_evals::ae-153-andrej-karpathy-on-evals", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-153-andrej-karpathy-on-evals::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-153-andrej-karpathy-on-evals", "source_section": "6 · Benchmark vs. eval (and benchmark integrity: contamination, saturation, label errors, leaderboard gaming)", "source_title": "Andrej Karpathy on evals", "source_type": "web_article", "source_url": "https://x.com/karpathy/status/1896266683301659068"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://arxiv.org/abs/2405.00332"], "content_sha256": "b99d31f9cad364ff7fb30f77e3cbee2046b316aa07071df90bdcb886f33da26f", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-154-a-careful-examination-of-llm-performance-on-grad.raw.txt", "primary_url": "https://arxiv.org/abs/2405.00332"}, "exemplar_id": "awesome_evals::ae-154-a-careful-examination-of-llm-performance-on-grad", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-154-a-careful-examination-of-llm-performance-on-grad::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-154-a-careful-examination-of-llm-performance-on-grad", "source_section": "6 · Benchmark vs. eval (and benchmark integrity: contamination, saturation, label errors, leaderboard gaming)", "source_title": "A Careful Examination of LLM Performance on Grade School Arithmetic (GSM1k)", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2405.00332"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://arxiv.org/abs/2103.14749"], "content_sha256": "75e133664f0918aaf814c24f68cbbde17b19522246a86410b03aa47a475a1659", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/benchmark-label-errors.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-155-pervasive-label-errors-in-test-sets-destabilize.raw.txt", "primary_url": "https://arxiv.org/abs/2103.14749"}, "exemplar_id": "awesome_evals::ae-155-pervasive-label-errors-in-test-sets-destabilize", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-155-pervasive-label-errors-in-test-sets-destabilize::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-155-pervasive-label-errors-in-test-sets-destabilize", "source_section": "6 · Benchmark vs. eval (and benchmark integrity: contamination, saturation, label errors, leaderboard gaming)", "source_title": "Pervasive Label Errors in Test Sets Destabilize Machine Learning Benchmarks", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2103.14749"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://arxiv.org/abs/2406.04127"], "content_sha256": "4a6ff40c77d2036aec987627e213c124ae348016b0f141448f205ced8b97b424", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-156-are-we-done-with-mmlu-mmlu-redux.raw.txt", "primary_url": "https://arxiv.org/abs/2406.04127"}, "exemplar_id": "awesome_evals::ae-156-are-we-done-with-mmlu-mmlu-redux", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-156-are-we-done-with-mmlu-mmlu-redux::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-156-are-we-done-with-mmlu-mmlu-redux", "source_section": "6 · Benchmark vs. eval (and benchmark integrity: contamination, saturation, label errors, leaderboard gaming)", "source_title": "Are We Done with MMLU? (MMLU-Redux)", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2406.04127"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://arxiv.org/abs/2403.07974"], "content_sha256": "c394e902fccb4705dcfcf3df5aa08c4376d19ef202a8879faff1107232333720", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-157-livecodebench-holistic-and-contamination-free-ev.raw.txt", "primary_url": "https://arxiv.org/abs/2403.07974"}, "exemplar_id": "awesome_evals::ae-157-livecodebench-holistic-and-contamination-free-ev", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-157-livecodebench-holistic-and-contamination-free-ev::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-157-livecodebench-holistic-and-contamination-free-ev", "source_section": "6 · Benchmark vs. eval (and benchmark integrity: contamination, saturation, label errors, leaderboard gaming)", "source_title": "LiveCodeBench: Holistic and Contamination-Free Evaluation of LLMs for Code", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2403.07974"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://github.com/LiveBench/LiveBench"], "content_sha256": "3b1845679ec5f530222bae9e702e6add1f6e14a7855674070332237ffbf9e5ee", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-158-livebench-a-challenging-contamination-limited-ll.raw.txt", "primary_url": "https://github.com/LiveBench/LiveBench"}, "exemplar_id": "awesome_evals::ae-158-livebench-a-challenging-contamination-limited-ll", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-158-livebench-a-challenging-contamination-limited-ll::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-158-livebench-a-challenging-contamination-limited-ll", "source_section": "6 · Benchmark vs. eval (and benchmark integrity: contamination, saturation, label errors, leaderboard gaming)", "source_title": "LiveBench: A Challenging, Contamination-Limited LLM Benchmark", "source_type": "repository_or_docs", "source_url": "https://github.com/LiveBench/LiveBench"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://github.com/huggingface/evaluation-guidebook"], "content_sha256": "6d7a762f14c291f97b6d8d06c0409274560472dabcf7226d722d2b8f038abc07", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-159-the-llm-evaluation-guidebook-open-llm-leaderboar.raw.txt", "primary_url": "https://github.com/huggingface/evaluation-guidebook"}, "exemplar_id": "awesome_evals::ae-159-the-llm-evaluation-guidebook-open-llm-leaderboar", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-159-the-llm-evaluation-guidebook-open-llm-leaderboar::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-159-the-llm-evaluation-guidebook-open-llm-leaderboar", "source_section": "6 · Benchmark vs. eval (and benchmark integrity: contamination, saturation, label errors, leaderboard gaming)", "source_title": "The LLM Evaluation Guidebook (Open LLM Leaderboard team)", "source_type": "repository_or_docs", "source_url": "https://github.com/huggingface/evaluation-guidebook"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://arxiv.org/abs/2510.11977"], "content_sha256": "350ef898cf10e85859147e9794539cc6f178cb70c3f383dd355cc337b2548919", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/langfuse-agent-evaluation-guide.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-160-holistic-agent-leaderboard-the-missing-infrastru.raw.txt", "primary_url": "https://arxiv.org/abs/2510.11977"}, "exemplar_id": "awesome_evals::ae-160-holistic-agent-leaderboard-the-missing-infrastru", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-160-holistic-agent-leaderboard-the-missing-infrastru::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-160-holistic-agent-leaderboard-the-missing-infrastru", "source_section": "6 · Benchmark vs. eval (and benchmark integrity: contamination, saturation, label errors, leaderboard gaming)", "source_title": "Holistic Agent Leaderboard: The Missing Infrastructure for AI Agent Evaluation", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2510.11977"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://blog.collinear.ai/p/gaming-the-system-goodharts-law-exemplified-in-ai-leaderboard-controversy"], "content_sha256": "609e517fb9faf0fbb90575b2f734978585a707145b860a8fce879bd1791ac052", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-161-gaming-the-system-goodhart-s-law-exemplified-in.raw.txt", "primary_url": "https://blog.collinear.ai/p/gaming-the-system-goodharts-law-exemplified-in-ai-leaderboard-controversy"}, "exemplar_id": "awesome_evals::ae-161-gaming-the-system-goodhart-s-law-exemplified-in", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-161-gaming-the-system-goodhart-s-law-exemplified-in::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-161-gaming-the-system-goodhart-s-law-exemplified-in", "source_section": "6 · Benchmark vs. eval (and benchmark integrity: contamination, saturation, label errors, leaderboard gaming)", "source_title": "Gaming the System: Goodhart's Law Exemplified in the AI Leaderboard Controversy", "source_type": "blog", "source_url": "https://blog.collinear.ai/p/gaming-the-system-goodharts-law-exemplified-in-ai-leaderboard-controversy"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://openai.com/index/trustworthy-third-party-evaluations-foundations/"], "content_sha256": "25e903ef685304e38a3b6c3928369efb510e714f6e0f9647fc709fb65c0488da", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-162-a-shared-playbook-for-trustworthy-third-party-ev.raw.txt", "primary_url": "https://openai.com/index/trustworthy-third-party-evaluations-foundations/"}, "exemplar_id": "awesome_evals::ae-162-a-shared-playbook-for-trustworthy-third-party-ev", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-162-a-shared-playbook-for-trustworthy-third-party-ev::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-162-a-shared-playbook-for-trustworthy-third-party-ev", "source_section": "6 · Benchmark vs. eval (and benchmark integrity: contamination, saturation, label errors, leaderboard gaming)", "source_title": "A Shared Playbook for Trustworthy Third-Party Evaluations", "source_type": "docs_or_book", "source_url": "https://openai.com/index/trustworthy-third-party-evaluations-foundations/"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2403.13787"], "content_sha256": "a8176cbcdef3902d01a231f9274dc475ad11a672eec995c1d7af5fa5b588a3a2", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-163-rewardbench.raw.txt", "primary_url": "https://arxiv.org/abs/2403.13787"}, "exemplar_id": "awesome_evals::ae-163-rewardbench", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-163-rewardbench::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-163-rewardbench", "source_section": "7 · Evals & RL environments (verifiers, reward design, difficulty calibration, lifecycle)", "source_title": "RewardBench", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2403.13787"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://www.interconnects.ai/p/the-new-rl-scaling-laws", "https://www.latent.space/p/the-rlvr-revolution-with-nathan-lambert"], "content_sha256": "1f61a39dd61add7266f332fbdba7c895263b495c6cf72b03418f291efbdaa5d8", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-164-the-new-rl-scaling-laws.raw.txt", "primary_url": "https://www.interconnects.ai/p/the-new-rl-scaling-laws"}, "exemplar_id": "awesome_evals::ae-164-the-new-rl-scaling-laws", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-164-the-new-rl-scaling-laws::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-164-the-new-rl-scaling-laws", "source_section": "7 · Evals & RL environments (verifiers, reward design, difficulty calibration, lifecycle)", "source_title": "The New RL Scaling Laws", "source_type": "blog", "source_url": "https://www.interconnects.ai/p/the-new-rl-scaling-laws"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2506.10947"], "content_sha256": "1f44ca6d14cf09da3e7387f28fdf85d53c45caa9eefa9220dede6356c0744341", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-165-spurious-rewards-rethinking-training-signals-in.raw.txt", "primary_url": "https://arxiv.org/abs/2506.10947"}, "exemplar_id": "awesome_evals::ae-165-spurious-rewards-rethinking-training-signals-in", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-165-spurious-rewards-rethinking-training-signals-in::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-165-spurious-rewards-rethinking-training-signals-in", "source_section": "7 · Evals & RL environments (verifiers, reward design, difficulty calibration, lifecycle)", "source_title": "Spurious Rewards: Rethinking Training Signals in RLVR", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2506.10947"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://www.interconnects.ai/p/the-state-of-post-training-2025"], "content_sha256": "09bf6865b676d05474ad3b8cf076fe702674f41163e88008e97dd7b94951fbee", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-166-the-state-of-post-training-2025.raw.txt", "primary_url": "https://www.interconnects.ai/p/the-state-of-post-training-2025"}, "exemplar_id": "awesome_evals::ae-166-the-state-of-post-training-2025", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-166-the-state-of-post-training-2025::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-166-the-state-of-post-training-2025", "source_section": "7 · Evals & RL environments (verifiers, reward design, difficulty calibration, lifecycle)", "source_title": "The State of Post-Training 2025", "source_type": "blog", "source_url": "https://www.interconnects.ai/p/the-state-of-post-training-2025"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://lilianweng.github.io/posts/2024-11-28-reward-hacking/"], "content_sha256": "99da73635a1800451d546644414161325c9fcd1455e971df5bf3dce987bbdd4e", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/countdown-code-reward-hacking-rlvr-testbed.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-167-reward-hacking-in-reinforcement-learning.raw.txt", "primary_url": "https://lilianweng.github.io/posts/2024-11-28-reward-hacking/"}, "exemplar_id": "awesome_evals::ae-167-reward-hacking-in-reinforcement-learning", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-167-reward-hacking-in-reinforcement-learning::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-167-reward-hacking-in-reinforcement-learning", "source_section": "7 · Evals & RL environments (verifiers, reward design, difficulty calibration, lifecycle)", "source_title": "Reward Hacking in Reinforcement Learning", "source_type": "web_article", "source_url": "https://lilianweng.github.io/posts/2024-11-28-reward-hacking/"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://deepmind.google/blog/specification-gaming-the-flip-side-of-ai-ingenuity/"], "content_sha256": "3e6df14ab8426ce1030836a5490f745dd81b8ed57396f269623fc0135c56adf5", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/recontextualization-mitigates-specification-gaming.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-168-specification-gaming-the-flip-side-of-ai-ingenui.raw.txt", "primary_url": "https://deepmind.google/blog/specification-gaming-the-flip-side-of-ai-ingenuity/"}, "exemplar_id": "awesome_evals::ae-168-specification-gaming-the-flip-side-of-ai-ingenui", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-168-specification-gaming-the-flip-side-of-ai-ingenui::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-168-specification-gaming-the-flip-side-of-ai-ingenui", "source_section": "7 · Evals & RL environments (verifiers, reward design, difficulty calibration, lifecycle)", "source_title": "Specification gaming: the flip side of AI ingenuity", "source_type": "blog", "source_url": "https://deepmind.google/blog/specification-gaming-the-flip-side-of-ai-ingenuity/"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://www.latent.space/p/willccbb"], "content_sha256": "9bceac71c76cadf193cabce5e616a89a0b251898326485c3ea3a5c7e24ae0ee3", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/florian-brand-prime-intellect-llm-benchmarks-era-of-agents.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-169-multi-turn-rl-for-multi-hour-agents-with-will-br.raw.txt", "primary_url": "https://www.latent.space/p/willccbb"}, "exemplar_id": "awesome_evals::ae-169-multi-turn-rl-for-multi-hour-agents-with-will-br", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-169-multi-turn-rl-for-multi-hour-agents-with-will-br::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-169-multi-turn-rl-for-multi-hour-agents-with-will-br", "source_section": "7 · Evals & RL environments (verifiers, reward design, difficulty calibration, lifecycle)", "source_title": "Multi-Turn RL for Multi-Hour Agents — with Will Brown (Prime Intellect)", "source_type": "blog", "source_url": "https://www.latent.space/p/willccbb"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2509.21882"], "content_sha256": "a91583004d519fa2d2b3859ea0ab40e28161cf4e95a0cee64682c23e55054f7a", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/rlvr-hidden-costs-measurement-gaps.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-170-position-the-hidden-costs-and-measurement-gaps-o.raw.txt", "primary_url": "https://arxiv.org/abs/2509.21882"}, "exemplar_id": "awesome_evals::ae-170-position-the-hidden-costs-and-measurement-gaps-o", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-170-position-the-hidden-costs-and-measurement-gaps-o::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-170-position-the-hidden-costs-and-measurement-gaps-o", "source_section": "7 · Evals & RL environments (verifiers, reward design, difficulty calibration, lifecycle)", "source_title": "Position: The Hidden Costs and Measurement Gaps of RLVR", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2509.21882"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2506.01937"], "content_sha256": "4d94b4cdada2a6b78fa35cc91abf450625023b26ad9bd0435a9c5af0d4f2c1c0", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-171-rewardbench-2-advancing-reward-model-evaluation.raw.txt", "primary_url": "https://arxiv.org/abs/2506.01937"}, "exemplar_id": "awesome_evals::ae-171-rewardbench-2-advancing-reward-model-evaluation", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-171-rewardbench-2-advancing-reward-model-evaluation::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-171-rewardbench-2-advancing-reward-model-evaluation", "source_section": "7 · Evals & RL environments (verifiers, reward design, difficulty calibration, lifecycle)", "source_title": "RewardBench 2: Advancing Reward Model Evaluation", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2506.01937"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://rlhfbook.com/c/05-reward-models"], "content_sha256": "773504c8e4b847899a4281e7f0ea03c1ac4caf68a6661ff63b5d6e7d9c94c051", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/papers/scalable-agent-alignment-via-reward-modeling.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-172-reward-modeling-rlhf-book-ch-5.raw.txt", "primary_url": "https://rlhfbook.com/c/05-reward-models"}, "exemplar_id": "awesome_evals::ae-172-reward-modeling-rlhf-book-ch-5", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-172-reward-modeling-rlhf-book-ch-5::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-172-reward-modeling-rlhf-book-ch-5", "source_section": "7 · Evals & RL environments (verifiers, reward design, difficulty calibration, lifecycle)", "source_title": "Reward Modeling (RLHF Book, ch. 5)", "source_type": "docs_or_book", "source_url": "https://rlhfbook.com/c/05-reward-models"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2506.06632"], "content_sha256": "3482bebffe22409a49d9cf94041f032893571985d1dc6503ceb1560b00445e11", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-173-curriculum-rl-from-easy-to-hard-tasks-improves-l.raw.txt", "primary_url": "https://arxiv.org/abs/2506.06632"}, "exemplar_id": "awesome_evals::ae-173-curriculum-rl-from-easy-to-hard-tasks-improves-l", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-173-curriculum-rl-from-easy-to-hard-tasks-improves-l::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-173-curriculum-rl-from-easy-to-hard-tasks-improves-l", "source_section": "7 · Evals & RL environments (verifiers, reward design, difficulty calibration, lifecycle)", "source_title": "Curriculum RL from Easy to Hard Tasks Improves LLM Reasoning (E2H Reasoner)", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2506.06632"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2512.19682"], "content_sha256": "a9c9cf9ad1246b964ecc264d46b2756b00de4f55de78c954d664d3b69bb432db", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-174-genenv-difficulty-aligned-co-evolution-between-l.raw.txt", "primary_url": "https://arxiv.org/abs/2512.19682"}, "exemplar_id": "awesome_evals::ae-174-genenv-difficulty-aligned-co-evolution-between-l", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-174-genenv-difficulty-aligned-co-evolution-between-l::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-174-genenv-difficulty-aligned-co-evolution-between-l", "source_section": "7 · Evals & RL environments (verifiers, reward design, difficulty calibration, lifecycle)", "source_title": "GenEnv: Difficulty-Aligned Co-Evolution Between LLM Agents and Environment Simulators", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2512.19682"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://eugeneyan.com/writing/llm-evaluators/"], "content_sha256": "7e856220e7e37db84c43ae6f109f6626323f094ef95f883cbfc610d9454b2511", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-175-evaluating-the-effectiveness-of-llm-evaluators.raw.txt", "primary_url": "https://eugeneyan.com/writing/llm-evaluators/"}, "exemplar_id": "awesome_evals::ae-175-evaluating-the-effectiveness-of-llm-evaluators", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-175-evaluating-the-effectiveness-of-llm-evaluators::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-175-evaluating-the-effectiveness-of-llm-evaluators", "source_section": "8 · LLM-as-judge & verifiers (alignment, biases, verifiable vs judgeable)", "source_title": "Evaluating the Effectiveness of LLM-Evaluators", "source_type": "web_article", "source_url": "https://eugeneyan.com/writing/llm-evaluators/"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://hamel.dev/blog/posts/llm-judge/"], "content_sha256": "f0f10e4fc7908bfee6e4452e1f64d6b711c9c962db00fbe51d7b36f7a3c43f8a", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-176-creating-an-llm-as-a-judge-that-drives-business.raw.txt", "primary_url": "https://hamel.dev/blog/posts/llm-judge/"}, "exemplar_id": "awesome_evals::ae-176-creating-an-llm-as-a-judge-that-drives-business", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-176-creating-an-llm-as-a-judge-that-drives-business::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-176-creating-an-llm-as-a-judge-that-drives-business", "source_section": "8 · LLM-as-judge & verifiers (alignment, biases, verifiable vs judgeable)", "source_title": "Creating an LLM-as-a-Judge That Drives Business Results", "source_type": "blog", "source_url": "https://hamel.dev/blog/posts/llm-judge/"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2404.12272", "https://people.eecs.berkeley.edu/~bjoern/papers/shankar-validators-uist2024.pdf"], "content_sha256": "2a98cac4c82e1b225ab4c6468262c9245923882e276c6f518b0142cc02ecae5a", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-177-who-validates-the-validators-evalgen.raw.txt", "primary_url": "https://arxiv.org/abs/2404.12272"}, "exemplar_id": "awesome_evals::ae-177-who-validates-the-validators-evalgen", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-177-who-validates-the-validators-evalgen::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-177-who-validates-the-validators-evalgen", "source_section": "8 · LLM-as-judge & verifiers (alignment, biases, verifiable vs judgeable)", "source_title": "Who Validates the Validators? (EvalGen)", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2404.12272"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://hamel.dev/blog/posts/evals-faq/"], "content_sha256": "b5d5398f91d39542cc52d6c11bc38dfb86e4da2d06fcb73e69add860602d8e02", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-178-llm-evals-faq.raw.txt", "primary_url": "https://hamel.dev/blog/posts/evals-faq/"}, "exemplar_id": "awesome_evals::ae-178-llm-evals-faq", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-178-llm-evals-faq::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-178-llm-evals-faq", "source_section": "8 · LLM-as-judge & verifiers (alignment, biases, verifiable vs judgeable)", "source_title": "LLM Evals FAQ", "source_type": "blog", "source_url": "https://hamel.dev/blog/posts/evals-faq/"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://leehanchung.github.io/blogs/2024/08/11/llm-as-a-judge/"], "content_sha256": "445735aa534efe43630d6513c3f67cf6a90d4369aa282e4b047624084a8575ca", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-179-llm-as-a-judge-rethinking-model-based-evaluation.raw.txt", "primary_url": "https://leehanchung.github.io/blogs/2024/08/11/llm-as-a-judge/"}, "exemplar_id": "awesome_evals::ae-179-llm-as-a-judge-rethinking-model-based-evaluation", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-179-llm-as-a-judge-rethinking-model-based-evaluation::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-179-llm-as-a-judge-rethinking-model-based-evaluation", "source_section": "8 · LLM-as-judge & verifiers (alignment, biases, verifiable vs judgeable)", "source_title": "LLM-as-a-Judge: Rethinking Model-Based Evaluations", "source_type": "blog", "source_url": "https://leehanchung.github.io/blogs/2024/08/11/llm-as-a-judge/"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2306.05685"], "content_sha256": "21da0bd112b6f623aa04437301b85ad4e581278b0b24e012932ea6abb2452808", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-pod-gd-gonzalez-chatbot-arena.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-180-judging-llm-as-a-judge-with-mt-bench-and-chatbot.raw.txt", "primary_url": "https://arxiv.org/abs/2306.05685"}, "exemplar_id": "awesome_evals::ae-180-judging-llm-as-a-judge-with-mt-bench-and-chatbot", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-180-judging-llm-as-a-judge-with-mt-bench-and-chatbot::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-180-judging-llm-as-a-judge-with-mt-bench-and-chatbot", "source_section": "8 · LLM-as-judge & verifiers (alignment, biases, verifiable vs judgeable)", "source_title": "Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2306.05685"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2406.18403"], "content_sha256": "c20a78d7515ed808d5371ce1b6e4dfd1756897bfc3e283375a6d42e523223f4b", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-yan-llms-as-judges.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-181-llms-instead-of-human-judges-a-large-scale-study.raw.txt", "primary_url": "https://arxiv.org/abs/2406.18403"}, "exemplar_id": "awesome_evals::ae-181-llms-instead-of-human-judges-a-large-scale-study", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-181-llms-instead-of-human-judges-a-large-scale-study::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-181-llms-instead-of-human-judges-a-large-scale-study", "source_section": "8 · LLM-as-judge & verifiers (alignment, biases, verifiable vs judgeable)", "source_title": "LLMs Instead of Human Judges? A Large-Scale Study", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2406.18403"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://eugeneyan.com/writing/aligneval/"], "content_sha256": "2b116b69d03ce1fd0148941be90ed71d5358008ad5fcb534defe066f23bb67be", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-182-aligneval.raw.txt", "primary_url": "https://eugeneyan.com/writing/aligneval/"}, "exemplar_id": "awesome_evals::ae-182-aligneval", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-182-aligneval::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-182-aligneval", "source_section": "8 · LLM-as-judge & verifiers (alignment, biases, verifiable vs judgeable)", "source_title": "AlignEval", "source_type": "web_article", "source_url": "https://eugeneyan.com/writing/aligneval/"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://eugeneyan.com/writing/product-evals/"], "content_sha256": "e4fc4af54f2781c8e534609e264181d69c2e648bf4b39e2e31b909afba569816", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-183-product-evals-in-three-simple-steps.raw.txt", "primary_url": "https://eugeneyan.com/writing/product-evals/"}, "exemplar_id": "awesome_evals::ae-183-product-evals-in-three-simple-steps", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-183-product-evals-in-three-simple-steps::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-183-product-evals-in-three-simple-steps", "source_section": "8 · LLM-as-judge & verifiers (alignment, biases, verifiable vs judgeable)", "source_title": "Product Evals in Three Simple Steps", "source_type": "web_article", "source_url": "https://eugeneyan.com/writing/product-evals/"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://leehanchung.github.io/blogs/2025/03/03/cohen-kappa/"], "content_sha256": "a0b67556e7644e0c207e6aa53941c5760325dba991b55f496ce4276f2000d78b", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-184-statistics-for-ai-ml-part-3-cohen-s-kappa.raw.txt", "primary_url": "https://leehanchung.github.io/blogs/2025/03/03/cohen-kappa/"}, "exemplar_id": "awesome_evals::ae-184-statistics-for-ai-ml-part-3-cohen-s-kappa", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-184-statistics-for-ai-ml-part-3-cohen-s-kappa::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-184-statistics-for-ai-ml-part-3-cohen-s-kappa", "source_section": "8 · LLM-as-judge & verifiers (alignment, biases, verifiable vs judgeable)", "source_title": "Statistics for AI/ML, Part 3 — Cohen's Kappa", "source_type": "blog", "source_url": "https://leehanchung.github.io/blogs/2025/03/03/cohen-kappa/"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://www.sh-reya.com/blog/ai-engineering-flywheel/"], "content_sha256": "b30c2c6170dab30e512d96c87c8a12ce77a12031a7aa6ba01b87aa5c09ba3ca9", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-185-data-flywheels-for-llm-applications.raw.txt", "primary_url": "https://www.sh-reya.com/blog/ai-engineering-flywheel/"}, "exemplar_id": "awesome_evals::ae-185-data-flywheels-for-llm-applications", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-185-data-flywheels-for-llm-applications::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-185-data-flywheels-for-llm-applications", "source_section": "8 · LLM-as-judge & verifiers (alignment, biases, verifiable vs judgeable)", "source_title": "Data Flywheels for LLM Applications", "source_type": "blog", "source_url": "https://www.sh-reya.com/blog/ai-engineering-flywheel/"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/html/2401.03038v1", "https://arxiv.org/abs/2410.12189"], "content_sha256": "589e71a79dec3221328d704e3dbb6021ce1d9a69169682d5db7175f8b3bf2830", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-186-spade.raw.txt", "primary_url": "https://arxiv.org/html/2401.03038v1"}, "exemplar_id": "awesome_evals::ae-186-spade", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-186-spade::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-186-spade", "source_section": "8 · LLM-as-judge & verifiers (alignment, biases, verifiable vs judgeable)", "source_title": "SPADE", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/html/2401.03038v1"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2404.13076"], "content_sha256": "0a377fb71e55eb3d05796fe88e215f646596fa91a44f7b95dfcad73ce40df706", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-187-llm-evaluators-recognize-and-favor-their-own-gen.raw.txt", "primary_url": "https://arxiv.org/abs/2404.13076"}, "exemplar_id": "awesome_evals::ae-187-llm-evaluators-recognize-and-favor-their-own-gen", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-187-llm-evaluators-recognize-and-favor-their-own-gen::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-187-llm-evaluators-recognize-and-favor-their-own-gen", "source_section": "8 · LLM-as-judge & verifiers (alignment, biases, verifiable vs judgeable)", "source_title": "LLM Evaluators Recognize and Favor Their Own Generations", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2404.13076"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2303.16634"], "content_sha256": "f3719d1cfc352630e15cacdc50b2e9debbed572c8a2ac5ff5180531f30766b97", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/han-lee-evaluation-and-alignment-seminal-papers.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-188-g-eval-nlg-evaluation-using-gpt-4-with-better-hu.raw.txt", "primary_url": "https://arxiv.org/abs/2303.16634"}, "exemplar_id": "awesome_evals::ae-188-g-eval-nlg-evaluation-using-gpt-4-with-better-hu", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-188-g-eval-nlg-evaluation-using-gpt-4-with-better-hu::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-188-g-eval-nlg-evaluation-using-gpt-4-with-better-hu", "source_section": "8 · LLM-as-judge & verifiers (alignment, biases, verifiable vs judgeable)", "source_title": "G-Eval: NLG Evaluation using GPT-4 with Better Human Alignment", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2303.16634"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2411.15594"], "content_sha256": "846daf987c9f7c91d10f5ccc06d87b1ac16ed585198e5d9e0e2cde520b1604bd", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-189-a-survey-on-llm-as-a-judge.raw.txt", "primary_url": "https://arxiv.org/abs/2411.15594"}, "exemplar_id": "awesome_evals::ae-189-a-survey-on-llm-as-a-judge", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-189-a-survey-on-llm-as-a-judge::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-189-a-survey-on-llm-as-a-judge", "source_section": "8 · LLM-as-judge & verifiers (alignment, biases, verifiable vs judgeable)", "source_title": "A Survey on LLM-as-a-Judge", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2411.15594"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2507.08794"], "content_sha256": "091e765a3071d3a192f7d6efe796266eb7b78c6f4f1414005dec9ca70a07504d", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-190-one-token-to-fool-llm-as-a-judge.raw.txt", "primary_url": "https://arxiv.org/abs/2507.08794"}, "exemplar_id": "awesome_evals::ae-190-one-token-to-fool-llm-as-a-judge", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-190-one-token-to-fool-llm-as-a-judge::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-190-one-token-to-fool-llm-as-a-judge", "source_section": "8 · LLM-as-judge & verifiers (alignment, biases, verifiable vs judgeable)", "source_title": "One Token to Fool LLM-as-a-Judge", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2507.08794"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://hazyresearch.stanford.edu/blog/2025-06-18-weaver"], "content_sha256": "840677ec0216a69fce6d958586f020226a5debde36636e04dff5de2769607616", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/papers/bertscore-evaluating-text-generation-with-bert.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-191-weaver-closing-the-generation-verification-gap-w.raw.txt", "primary_url": "https://hazyresearch.stanford.edu/blog/2025-06-18-weaver"}, "exemplar_id": "awesome_evals::ae-191-weaver-closing-the-generation-verification-gap-w", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-191-weaver-closing-the-generation-verification-gap-w::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-191-weaver-closing-the-generation-verification-gap-w", "source_section": "8 · LLM-as-judge & verifiers (alignment, biases, verifiable vs judgeable)", "source_title": "Weaver: Closing the Generation-Verification Gap with Weak Verifiers", "source_type": "blog", "source_url": "https://hazyresearch.stanford.edu/blog/2025-06-18-weaver"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2410.10934"], "content_sha256": "a4ee22ac1aea42a45d3528cae187b90baf690619d29af3f63fc324c90b364a44", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/opik-evaluate-agent-trajectory.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-192-agent-as-a-judge-evaluate-agents-with-agents.raw.txt", "primary_url": "https://arxiv.org/abs/2410.10934"}, "exemplar_id": "awesome_evals::ae-192-agent-as-a-judge-evaluate-agents-with-agents", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-192-agent-as-a-judge-evaluate-agents-with-agents::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-192-agent-as-a-judge-evaluate-agents-with-agents", "source_section": "8 · LLM-as-judge & verifiers (alignment, biases, verifiable vs judgeable)", "source_title": "Agent-as-a-Judge: Evaluate Agents with Agents", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2410.10934"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2507.09884"], "content_sha256": "432199f64f3f479059f0abb8086549652d7493b74d3650be2bcd7765268ad80b", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-193-verifybench-a-systematic-benchmark-for-evaluatin.raw.txt", "primary_url": "https://arxiv.org/abs/2507.09884"}, "exemplar_id": "awesome_evals::ae-193-verifybench-a-systematic-benchmark-for-evaluatin", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-193-verifybench-a-systematic-benchmark-for-evaluatin::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-193-verifybench-a-systematic-benchmark-for-evaluatin", "source_section": "8 · LLM-as-judge & verifiers (alignment, biases, verifiable vs judgeable)", "source_title": "VerifyBench: A Systematic Benchmark for Evaluating Reasoning Verifiers Across Domains", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2507.09884"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://www.databricks.com/blog/pilot-production-custom-judges"], "content_sha256": "e41f65bac3a11644c1b819112201e0cffff08583f0590d28bbfc39458b8cfc35", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/godaddy-calibrating-llm-judge-scores.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-194-enhancing-llm-as-a-judge-with-grading-notes-from.raw.txt", "primary_url": "https://www.databricks.com/blog/pilot-production-custom-judges"}, "exemplar_id": "awesome_evals::ae-194-enhancing-llm-as-a-judge-with-grading-notes-from", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-194-enhancing-llm-as-a-judge-with-grading-notes-from::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-194-enhancing-llm-as-a-judge-with-grading-notes-from", "source_section": "8 · LLM-as-judge & verifiers (alignment, biases, verifiable vs judgeable)", "source_title": "Enhancing LLM-as-a-Judge with Grading Notes / From Pilot to Production with Custom Judges", "source_type": "blog", "source_url": "https://www.databricks.com/blog/pilot-production-custom-judges"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2410.02736"], "content_sha256": "777ce179d87154d6e0c810bc88dac2968b1a65706d8e01869a433b15d1070595", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-195-justice-or-prejudice-quantifying-biases-in-llm-a.raw.txt", "primary_url": "https://arxiv.org/abs/2410.02736"}, "exemplar_id": "awesome_evals::ae-195-justice-or-prejudice-quantifying-biases-in-llm-a", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-195-justice-or-prejudice-quantifying-biases-in-llm-a::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-195-justice-or-prejudice-quantifying-biases-in-llm-a", "source_section": "8 · LLM-as-judge & verifiers (alignment, biases, verifiable vs judgeable)", "source_title": "Justice or Prejudice? Quantifying Biases in LLM-as-a-Judge (CALM framework)", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2410.02736"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://www.anthropic.com/engineering/demystifying-evals-for-ai-agents"], "content_sha256": "5bbd04d4cbfd341cb3294b7da1bd7bbb56b5a0fefb3c1db81b81158de6066802", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/vanishing-gradients-agents-evals.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-196-demystifying-evals-for-ai-agents.raw.txt", "primary_url": "https://www.anthropic.com/engineering/demystifying-evals-for-ai-agents"}, "exemplar_id": "awesome_evals::ae-196-demystifying-evals-for-ai-agents", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-196-demystifying-evals-for-ai-agents::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-196-demystifying-evals-for-ai-agents", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "Demystifying Evals for AI Agents", "source_type": "web_article", "source_url": "https://www.anthropic.com/engineering/demystifying-evals-for-ai-agents"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2406.12045", "https://github.com/sierra-research/tau-bench"], "content_sha256": "20c2acc833d099fde9e8a4d9b7788e364a37d92f4ae727e332693fc5bc794034", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-197-bench-bench.raw.txt", "primary_url": "https://arxiv.org/abs/2406.12045"}, "exemplar_id": "awesome_evals::ae-197-bench-bench", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-197-bench-bench::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-197-bench-bench", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "τ-bench / τ²-bench", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2406.12045"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://sierra.ai/blog/benchmarking-ai-agents"], "content_sha256": "8fe932f38e8ac7d3297c478d7aadec32ef89e5dfc8589551865226e608b80e18", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-198-benchmarking-ai-agents.raw.txt", "primary_url": "https://sierra.ai/blog/benchmarking-ai-agents"}, "exemplar_id": "awesome_evals::ae-198-benchmarking-ai-agents", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-198-benchmarking-ai-agents::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-198-benchmarking-ai-agents", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "Benchmarking AI Agents", "source_type": "blog", "source_url": "https://sierra.ai/blog/benchmarking-ai-agents"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2311.12983"], "content_sha256": "d54599ae4bf96dac9b35e75ee9fc7693b9f73e0ac7d79610eb2ce0c7f69c4c0c", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-199-gaia-a-benchmark-for-general-ai-assistants.raw.txt", "primary_url": "https://arxiv.org/abs/2311.12983"}, "exemplar_id": "awesome_evals::ae-199-gaia-a-benchmark-for-general-ai-assistants", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-199-gaia-a-benchmark-for-general-ai-assistants::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-199-gaia-a-benchmark-for-general-ai-assistants", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "GAIA: A Benchmark for General AI Assistants", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2311.12983"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://eugeneyan.com/writing/cybersecurity-evals/"], "content_sha256": "e685d714fbb9e4b4e0ff95869ff71d1c2e29644099e9a3f3961dd74e15c67492", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-200-patterns-for-building-cybersecurity-evals.raw.txt", "primary_url": "https://eugeneyan.com/writing/cybersecurity-evals/"}, "exemplar_id": "awesome_evals::ae-200-patterns-for-building-cybersecurity-evals", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-200-patterns-for-building-cybersecurity-evals::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-200-patterns-for-building-cybersecurity-evals", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "Patterns for Building Cybersecurity Evals", "source_type": "web_article", "source_url": "https://eugeneyan.com/writing/cybersecurity-evals/"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://leehanchung.github.io/blogs/2025/09/08/pass-at-k/"], "content_sha256": "a1e2b09df956c3a8a6ffa148beae2f5e100a905ba7607ba7efa4bce66c804d7c", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-201-statistics-for-ai-ml-part-4-pass-k-and-unbiased.raw.txt", "primary_url": "https://leehanchung.github.io/blogs/2025/09/08/pass-at-k/"}, "exemplar_id": "awesome_evals::ae-201-statistics-for-ai-ml-part-4-pass-k-and-unbiased", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-201-statistics-for-ai-ml-part-4-pass-k-and-unbiased::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-201-statistics-for-ai-ml-part-4-pass-k-and-unbiased", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "Statistics for AI/ML, Part 4 — pass@k and Unbiased Estimator", "source_type": "blog", "source_url": "https://leehanchung.github.io/blogs/2025/09/08/pass-at-k/"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://leehanchung.github.io/blogs/2024/05/22/first-principles-eval/"], "content_sha256": "ef1e2ff8f19addeed18163cb3ef6120736d892a37691875ff65774ddd03ddfa3", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-202-first-principles-eval.raw.txt", "primary_url": "https://leehanchung.github.io/blogs/2024/05/22/first-principles-eval/"}, "exemplar_id": "awesome_evals::ae-202-first-principles-eval", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-202-first-principles-eval::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-202-first-principles-eval", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "First-Principles Eval", "source_type": "blog", "source_url": "https://leehanchung.github.io/blogs/2024/05/22/first-principles-eval/"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://github.com/SWE-bench/SWE-bench/blob/main/swebench/harness/grading.py", "https://swe-agent.com/0.7/background/aci/"], "content_sha256": "c573425751346905af1839c4788dee75ade837b380a7c20351abe09e51f8d373", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-203-swe-bench-grading-harness.raw.txt", "primary_url": "https://github.com/SWE-bench/SWE-bench/blob/main/swebench/harness/grading.py"}, "exemplar_id": "awesome_evals::ae-203-swe-bench-grading-harness", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-203-swe-bench-grading-harness::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-203-swe-bench-grading-harness", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "SWE-bench grading harness", "source_type": "repository_or_docs", "source_url": "https://github.com/SWE-bench/SWE-bench/blob/main/swebench/harness/grading.py"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://github.com/openai/human-eval/blob/master/human_eval/evaluation.py"], "content_sha256": "934c5577467c6e6b8373b37267ed050e4f2f5f09177f28f24b38abdabcec5d7e", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/dont-pass-at-k-bayesian-llm-eval.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-204-human-eval-pass-k-estimator.raw.txt", "primary_url": "https://github.com/openai/human-eval/blob/master/human_eval/evaluation.py"}, "exemplar_id": "awesome_evals::ae-204-human-eval-pass-k-estimator", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-204-human-eval-pass-k-estimator::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-204-human-eval-pass-k-estimator", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "human-eval (pass@k estimator)", "source_type": "repository_or_docs", "source_url": "https://github.com/openai/human-eval/blob/master/human_eval/evaluation.py"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2307.13854"], "content_sha256": "3e0932bb4b0bc38463632911d4186e15b09a4fc2b997457225912db251e4ced5", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-205-webarena-a-realistic-web-environment-for-buildin.raw.txt", "primary_url": "https://arxiv.org/abs/2307.13854"}, "exemplar_id": "awesome_evals::ae-205-webarena-a-realistic-web-environment-for-buildin", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-205-webarena-a-realistic-web-environment-for-buildin::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-205-webarena-a-realistic-web-environment-for-buildin", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "WebArena: A Realistic Web Environment for Building Autonomous Agents", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2307.13854"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2404.07972"], "content_sha256": "e77b1ae61fe55ed56204c0a4414344de02575d81d9cd7ec08d009ae4296ae7c6", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-206-osworld-benchmarking-multimodal-agents-for-open.raw.txt", "primary_url": "https://arxiv.org/abs/2404.07972"}, "exemplar_id": "awesome_evals::ae-206-osworld-benchmarking-multimodal-agents-for-open", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-206-osworld-benchmarking-multimodal-agents-for-open::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-206-osworld-benchmarking-multimodal-agents-for-open", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "OSWorld: Benchmarking Multimodal Agents for Open-Ended Tasks in Real Computer Environments", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2404.07972"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://www.tbench.ai/"], "content_sha256": "8e48e19a0568541bec63b2dcd5b37354cc85e92514550522f833f924126c167e", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-207-terminal-bench-benchmarking-agents-on-hard-reali.raw.txt", "primary_url": "https://www.tbench.ai/"}, "exemplar_id": "awesome_evals::ae-207-terminal-bench-benchmarking-agents-on-hard-reali", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-207-terminal-bench-benchmarking-agents-on-hard-reali::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-207-terminal-bench-benchmarking-agents-on-hard-reali", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command-Line Interfaces", "source_type": "web_article", "source_url": "https://www.tbench.ai/"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2408.08926"], "content_sha256": "237b1a798e38be5ea69bd9cfa3285a0eff54b1d37b8f78c7c23ef5c2326c9365", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/papers/evaluating-large-language-models-trained-on-code.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-208-cybench-a-framework-for-evaluating-cybersecurity.raw.txt", "primary_url": "https://arxiv.org/abs/2408.08926"}, "exemplar_id": "awesome_evals::ae-208-cybench-a-framework-for-evaluating-cybersecurity", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-208-cybench-a-framework-for-evaluating-cybersecurity::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-208-cybench-a-framework-for-evaluating-cybersecurity", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "Cybench: A Framework for Evaluating Cybersecurity Capabilities and Risk of Language Models", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2408.08926"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2504.08942"], "content_sha256": "c51c2880d33febc75d8ab9f8a042977449e2eb50c79eb5fde0d59789aa64105d", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/agentrewardbench-evaluating-automatic-evaluations-web-agent-.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-209-agentrewardbench-evaluating-automatic-evaluation.raw.txt", "primary_url": "https://arxiv.org/abs/2504.08942"}, "exemplar_id": "awesome_evals::ae-209-agentrewardbench-evaluating-automatic-evaluation", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-209-agentrewardbench-evaluating-automatic-evaluation::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-209-agentrewardbench-evaluating-automatic-evaluation", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "AgentRewardBench: Evaluating Automatic Evaluations of Web Agent Trajectories", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2504.08942"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2503.13657"], "content_sha256": "26ec58b035635eaa8f906ad94b9710219eadae3d454ea2cc7d1350ce41252905", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/mast-why-multi-agent-llm-systems-fail.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-210-why-do-multi-agent-llm-systems-fail-mast-taxonom.raw.txt", "primary_url": "https://arxiv.org/abs/2503.13657"}, "exemplar_id": "awesome_evals::ae-210-why-do-multi-agent-llm-systems-fail-mast-taxonom", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-210-why-do-multi-agent-llm-systems-fail-mast-taxonom::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-210-why-do-multi-agent-llm-systems-fail-mast-taxonom", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "Why Do Multi-Agent LLM Systems Fail? (MAST taxonomy)", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2503.13657"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://aclanthology.org/2024.acl-long.850/"], "content_sha256": "54c3267f4277f9d7ad04dcdf043c5213eb8ef274e7bb558df3614c3fccaf840d", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-211-appworld-a-controllable-world-of-apps-and-people.raw.txt", "primary_url": "https://aclanthology.org/2024.acl-long.850/"}, "exemplar_id": "awesome_evals::ae-211-appworld-a-controllable-world-of-apps-and-people", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-211-appworld-a-controllable-world-of-apps-and-people::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-211-appworld-a-controllable-world-of-apps-and-people", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "AppWorld: A Controllable World of Apps and People for Benchmarking Interactive Coding Agents", "source_type": "web_article", "source_url": "https://aclanthology.org/2024.acl-long.850/"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://openai.com/index/browsecomp/"], "content_sha256": "44aefab4e553fd8f2928eef6028f4c6cb2c282f5dd06e82d73b0d8515d9827e3", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-212-browsecomp-a-simple-yet-challenging-benchmark-fo.raw.txt", "primary_url": "https://openai.com/index/browsecomp/"}, "exemplar_id": "awesome_evals::ae-212-browsecomp-a-simple-yet-challenging-benchmark-fo", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-212-browsecomp-a-simple-yet-challenging-benchmark-fo::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-212-browsecomp-a-simple-yet-challenging-benchmark-fo", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "BrowseComp: A Simple Yet Challenging Benchmark for Browsing Agents", "source_type": "web_article", "source_url": "https://openai.com/index/browsecomp/"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2503.09089"], "content_sha256": "9a8a7def2c02b3c2a3f393ef9cc3b42aefa846b8ccb844a8d9c753fd3bf43c4d", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-213-locagent-graph-guided-llm-agents-for-code-locali.raw.txt", "primary_url": "https://arxiv.org/abs/2503.09089"}, "exemplar_id": "awesome_evals::ae-213-locagent-graph-guided-llm-agents-for-code-locali", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-213-locagent-graph-guided-llm-agents-for-code-locali::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-213-locagent-graph-guided-llm-agents-for-code-locali", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "LocAgent: Graph-Guided LLM Agents for Code Localization", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2503.09089"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2401.13919"], "content_sha256": "c50f32d9b606e0d9b4ea722bc9ce6474642ae43db52824e736c77a09c4b1ca46", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/papers/evaluating-large-language-models-trained-on-code.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-214-webvoyager-building-an-end-to-end-web-agent-with.raw.txt", "primary_url": "https://arxiv.org/abs/2401.13919"}, "exemplar_id": "awesome_evals::ae-214-webvoyager-building-an-end-to-end-web-agent-with", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-214-webvoyager-building-an-end-to-end-web-agent-with::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-214-webvoyager-building-an-end-to-end-web-agent-with", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "WebVoyager: Building an End-to-End Web Agent with Large Multimodal Models", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2401.13919"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://github.com/benchflow-ai/skillsbench"], "content_sha256": "67486d0bf73dbeeb321819a67a8da466a2a68302f23a525dc84564a1e4e61912", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-215-skillsbench.raw.txt", "primary_url": "https://github.com/benchflow-ai/skillsbench"}, "exemplar_id": "awesome_evals::ae-215-skillsbench", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-215-skillsbench::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-215-skillsbench", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "SkillsBench", "source_type": "repository_or_docs", "source_url": "https://github.com/benchflow-ai/skillsbench"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://github.com/benchflow-ai/ClawsBench"], "content_sha256": "c430b9b5d53cb0517d745980ee5f431649a54d9098be20b231323971261e60bb", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-216-clawsbench.raw.txt", "primary_url": "https://github.com/benchflow-ai/ClawsBench"}, "exemplar_id": "awesome_evals::ae-216-clawsbench", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-216-clawsbench::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-216-clawsbench", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "ClawsBench", "source_type": "repository_or_docs", "source_url": "https://github.com/benchflow-ai/ClawsBench"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://openai.com/index/introducing-swe-bench-verified/"], "content_sha256": "1686806c33aee99cf1ef603796a78398a52122df1461a7b37ac55462e1c66d3a", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-217-swe-bench-verified.raw.txt", "primary_url": "https://openai.com/index/introducing-swe-bench-verified/"}, "exemplar_id": "awesome_evals::ae-217-swe-bench-verified", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-217-swe-bench-verified::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-217-swe-bench-verified", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "SWE-bench Verified", "source_type": "web_article", "source_url": "https://openai.com/index/introducing-swe-bench-verified/"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2410.03859"], "content_sha256": "a0f30345dcc66858d3aaabd7843f9307c0357aaf89c0bc740588017b793c6544", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-218-swe-bench-multimodal.raw.txt", "primary_url": "https://arxiv.org/abs/2410.03859"}, "exemplar_id": "awesome_evals::ae-218-swe-bench-multimodal", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-218-swe-bench-multimodal::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-218-swe-bench-multimodal", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "SWE-bench Multimodal", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2410.03859"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2509.16941"], "content_sha256": "4a51b2c979fbd6c4565c870f98ad7110ab5f5930b3afcb2df2b798c4af02e69e", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-219-swe-bench-pro.raw.txt", "primary_url": "https://arxiv.org/abs/2509.16941"}, "exemplar_id": "awesome_evals::ae-219-swe-bench-pro", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-219-swe-bench-pro::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-219-swe-bench-pro", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "SWE-bench Pro", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2509.16941"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2502.12115"], "content_sha256": "31df4daa05d3fce12d5cdbd76889291ae7e10869f25c058e0bbd2a97ba3bfb2d", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-220-swe-lancer.raw.txt", "primary_url": "https://arxiv.org/abs/2502.12115"}, "exemplar_id": "awesome_evals::ae-220-swe-lancer", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-220-swe-lancer::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-220-swe-lancer", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "SWE-Lancer", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2502.12115"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2412.21139"], "content_sha256": "59c30f736957e4236bdddc7b5fa0b83cc173699e12a374629fe0cf71b0fd7f1d", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-221-swe-gym.raw.txt", "primary_url": "https://arxiv.org/abs/2412.21139"}, "exemplar_id": "awesome_evals::ae-221-swe-gym", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-221-swe-gym::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-221-swe-gym", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "SWE-Gym", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2412.21139"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2504.02605"], "content_sha256": "a0359689007891f8468eb7de76c85edcc7642c1fac8ce14b363dbd50a1c2c6cf", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-222-multi-swe-bench.raw.txt", "primary_url": "https://arxiv.org/abs/2504.02605"}, "exemplar_id": "awesome_evals::ae-222-multi-swe-bench", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-222-multi-swe-bench::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-222-multi-swe-bench", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "Multi-SWE-bench", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2504.02605"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2505.20411"], "content_sha256": "ba95da25e8f0fb10d72db26134b067e29b07092cf18ce172e432083860437a29", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-223-swe-rebench.raw.txt", "primary_url": "https://arxiv.org/abs/2505.20411"}, "exemplar_id": "awesome_evals::ae-223-swe-rebench", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-223-swe-rebench::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-223-swe-rebench", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "SWE-rebench", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2505.20411"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2411.15114"], "content_sha256": "d203ddc12bf2519b04154217f4f558e5021e9977e3b54bf047bae4e3e374c9ea", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-224-re-bench.raw.txt", "primary_url": "https://arxiv.org/abs/2411.15114"}, "exemplar_id": "awesome_evals::ae-224-re-bench", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-224-re-bench::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-224-re-bench", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "RE-Bench", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2411.15114"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2410.07095", "https://github.com/openai/mle-bench"], "content_sha256": "006f3c4b166fbf8d4b7a6a4cf5f02a0284d67af3d6d3f44f6b6a105c9dc7d2ac", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-225-mle-bench.raw.txt", "primary_url": "https://arxiv.org/abs/2410.07095"}, "exemplar_id": "awesome_evals::ae-225-mle-bench", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-225-mle-bench::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-225-mle-bench", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "MLE-bench", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2410.07095"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2504.01848"], "content_sha256": "3a4ef4172eb5fc3328c8d6f93c2075177d3518563dcfed2c4745c3bf119b7de3", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-226-paperbench.raw.txt", "primary_url": "https://arxiv.org/abs/2504.01848"}, "exemplar_id": "awesome_evals::ae-226-paperbench", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-226-paperbench::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-226-paperbench", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "PaperBench", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2504.01848"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://www.kaggle.com/competitions/konwinski-prize"], "content_sha256": "c950ee0f8f2749e1d218f0419e484058a090d5d8fd485336ec85e75764b0417f", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-227-konwinski-prize-k-prize.raw.txt", "primary_url": "https://www.kaggle.com/competitions/konwinski-prize"}, "exemplar_id": "awesome_evals::ae-227-konwinski-prize-k-prize", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-227-konwinski-prize-k-prize::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-227-konwinski-prize-k-prize", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "Konwinski Prize (K Prize)", "source_type": "web_article", "source_url": "https://www.kaggle.com/competitions/konwinski-prize"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2506.21506"], "content_sha256": "70e9d488d3300e02f0b141eed842da22a26bf58ba5584deab0bdd115c0da450b", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/agentrewardbench-evaluating-automatic-evaluations-web-agent-.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-228-mind2web-2-evaluating-agentic-search-with-agent.raw.txt", "primary_url": "https://arxiv.org/abs/2506.21506"}, "exemplar_id": "awesome_evals::ae-228-mind2web-2-evaluating-agentic-search-with-agent", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-228-mind2web-2-evaluating-agentic-search-with-agent::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-228-mind2web-2-evaluating-agentic-search-with-agent", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "Mind2Web 2: Evaluating Agentic Search with Agent-as-a-Judge", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2506.21506"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2504.01382"], "content_sha256": "101a54f3aec22cd1cc7359b7621ddcf862670c10b0c5f0a459e41de080ec0161", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-229-online-mind2web-an-illusion-of-progress-assessin.raw.txt", "primary_url": "https://arxiv.org/abs/2504.01382"}, "exemplar_id": "awesome_evals::ae-229-online-mind2web-an-illusion-of-progress-assessin", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-229-online-mind2web-an-illusion-of-progress-assessin::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-229-online-mind2web-an-illusion-of-progress-assessin", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "Online-Mind2Web (An Illusion of Progress? Assessing the Current State of Web Agents)", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2504.01382"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://github.com/agi-inc/REAL"], "content_sha256": "935700e768458c70d4ecb566191bf00ca4e662d1c36dbfa0c6ad89bdeb8001e3", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-230-real-benchmarking-autonomous-agents-on-determini.raw.txt", "primary_url": "https://github.com/agi-inc/REAL"}, "exemplar_id": "awesome_evals::ae-230-real-benchmarking-autonomous-agents-on-determini", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-230-real-benchmarking-autonomous-agents-on-determini::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-230-real-benchmarking-autonomous-agents-on-determini", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "REAL: Benchmarking Autonomous Agents on Deterministic Simulations of Real Websites", "source_type": "repository_or_docs", "source_url": "https://github.com/agi-inc/REAL"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2502.18356"], "content_sha256": "d3c727848f9ee6c49829e9a182711c82b3bdfab6356279b0784aaee723a6d755", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-231-webgames-challenging-general-purpose-web-browsin.raw.txt", "primary_url": "https://arxiv.org/abs/2502.18356"}, "exemplar_id": "awesome_evals::ae-231-webgames-challenging-general-purpose-web-browsin", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-231-webgames-challenging-general-purpose-web-browsin::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-231-webgames-challenging-general-purpose-web-browsin", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "WebGames: Challenging General-Purpose Web-Browsing AI Agents", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2502.18356"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://gorilla.cs.berkeley.edu/leaderboard.html"], "content_sha256": "ef72a8621b6ea1b8aa1d1b7691f73cc3079fd38e86f599c9a38cb626c911fc29", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-232-berkeley-function-calling-leaderboard-bfcl-v4.raw.txt", "primary_url": "https://gorilla.cs.berkeley.edu/leaderboard.html"}, "exemplar_id": "awesome_evals::ae-232-berkeley-function-calling-leaderboard-bfcl-v4", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-232-berkeley-function-calling-leaderboard-bfcl-v4::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-232-berkeley-function-calling-leaderboard-bfcl-v4", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "Berkeley Function Calling Leaderboard (BFCL) V4", "source_type": "web_article", "source_url": "https://gorilla.cs.berkeley.edu/leaderboard.html"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2407.08713"], "content_sha256": "5a453addff83aecf3bd024b8751daa866aee12515349191f85f8cefee9af8d4b", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-233-gta-a-benchmark-for-general-tool-agents.raw.txt", "primary_url": "https://arxiv.org/abs/2407.08713"}, "exemplar_id": "awesome_evals::ae-233-gta-a-benchmark-for-general-tool-agents", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-233-gta-a-benchmark-for-general-tool-agents::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-233-gta-a-benchmark-for-general-tool-agents", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "GTA: A Benchmark for General Tool Agents", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2407.08713"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2411.07763"], "content_sha256": "7e59f56d54f87378b91776b84c91a681aa4459571f60895ba126b9ec4b123e36", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/papers/evaluating-large-language-models-trained-on-code.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-234-spider-2-0-evaluating-language-models-on-real-wo.raw.txt", "primary_url": "https://arxiv.org/abs/2411.07763"}, "exemplar_id": "awesome_evals::ae-234-spider-2-0-evaluating-language-models-on-real-wo", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-234-spider-2-0-evaluating-language-models-on-real-wo::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-234-spider-2-0-evaluating-language-models-on-real-wo", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "Spider 2.0: Evaluating Language Models on Real-World Enterprise Text-to-SQL Workflows", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2411.07763"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2405.14573"], "content_sha256": "60d870e8b18941976df40efc51053cb8cf4d12d80d6350382a20abe5388e7be4", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-235-androidworld-a-dynamic-benchmarking-environment.raw.txt", "primary_url": "https://arxiv.org/abs/2405.14573"}, "exemplar_id": "awesome_evals::ae-235-androidworld-a-dynamic-benchmarking-environment", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-235-androidworld-a-dynamic-benchmarking-environment::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-235-androidworld-a-dynamic-benchmarking-environment", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "AndroidWorld: A Dynamic Benchmarking Environment for Autonomous Agents", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2405.14573"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2409.08264"], "content_sha256": "84d98bf0989464088288176255c44f4d8626513137181d78adad11ac1e45b990", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/aws-evaluating-ai-agents-amazon.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-236-windowsagentarena-evaluating-multi-modal-os-agen.raw.txt", "primary_url": "https://arxiv.org/abs/2409.08264"}, "exemplar_id": "awesome_evals::ae-236-windowsagentarena-evaluating-multi-modal-os-agen", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-236-windowsagentarena-evaluating-multi-modal-os-agen::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-236-windowsagentarena-evaluating-multi-modal-os-agen", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "WindowsAgentArena: Evaluating Multi-Modal OS Agents at Scale", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2409.08264"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2410.06703"], "content_sha256": "f8cb195a8207f0e600d9e57258316c1d4339f03014b8a47a9f74bba8af13ba66", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/aws-evaluating-ai-agents-amazon.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-237-st-webagentbench-evaluating-safety-and-trustwort.raw.txt", "primary_url": "https://arxiv.org/abs/2410.06703"}, "exemplar_id": "awesome_evals::ae-237-st-webagentbench-evaluating-safety-and-trustwort", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-237-st-webagentbench-evaluating-safety-and-trustwort::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-237-st-webagentbench-evaluating-safety-and-trustwort", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "ST-WebAgentBench: Evaluating Safety and Trustworthiness in Web Agents", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2410.06703"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2412.14161"], "content_sha256": "6ba2eaee612d357a47fc644af934cee28c829c160dc8a467a5b4188134ce598e", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-238-theagentcompany-benchmarking-llm-agents-on-conse.raw.txt", "primary_url": "https://arxiv.org/abs/2412.14161"}, "exemplar_id": "awesome_evals::ae-238-theagentcompany-benchmarking-llm-agents-on-conse", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-238-theagentcompany-benchmarking-llm-agents-on-conse::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-238-theagentcompany-benchmarking-llm-agents-on-conse", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "TheAgentCompany: Benchmarking LLM Agents on Consequential Real World Tasks", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2412.14161"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2401.13649"], "content_sha256": "760c23e0402ec70c3e5374d3accb5a29dea325e15150a02032f00f34355b6fbd", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/aws-evaluating-ai-agents-amazon.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-239-visualwebarena-evaluating-multimodal-agents-on-r.raw.txt", "primary_url": "https://arxiv.org/abs/2401.13649"}, "exemplar_id": "awesome_evals::ae-239-visualwebarena-evaluating-multimodal-agents-on-r", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-239-visualwebarena-evaluating-multimodal-agents-on-r::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-239-visualwebarena-evaluating-multimodal-agents-on-r", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "VisualWebArena: Evaluating Multimodal Agents on Realistic Visual Web Tasks", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2401.13649"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2510.04374"], "content_sha256": "8fe58faa9b4bce0d8dbbe265069fa65384c7b1f63720411cdf95cbc04b449f21", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-240-gdpval-evaluating-ai-model-performance-on-real-w.raw.txt", "primary_url": "https://arxiv.org/abs/2510.04374"}, "exemplar_id": "awesome_evals::ae-240-gdpval-evaluating-ai-model-performance-on-real-w", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-240-gdpval-evaluating-ai-model-performance-on-real-w::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-240-gdpval-evaluating-ai-model-performance-on-real-w", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "GDPval: Evaluating AI Model Performance on Real-World Economically Valuable Tasks", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2510.04374"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2510.26787"], "content_sha256": "af08512d4b36ec998e87e619b6a674f0e810cad81e30cc48ab3309d6ffec030d", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-241-remote-labor-index-measuring-ai-automation-of-re.raw.txt", "primary_url": "https://arxiv.org/abs/2510.26787"}, "exemplar_id": "awesome_evals::ae-241-remote-labor-index-measuring-ai-automation-of-re", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-241-remote-labor-index-measuring-ai-automation-of-re::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-241-remote-labor-index-measuring-ai-automation-of-re", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "Remote Labor Index: Measuring AI Automation of Remote Work", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2510.26787"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2501.14249"], "content_sha256": "c61040e24b5fc0e81b7cb3813daf47c0b6fcf4cfff98531cc93ed2260b62f292", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-242-humanity-s-last-exam.raw.txt", "primary_url": "https://arxiv.org/abs/2501.14249"}, "exemplar_id": "awesome_evals::ae-242-humanity-s-last-exam", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-242-humanity-s-last-exam::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-242-humanity-s-last-exam", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "Humanity's Last Exam", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2501.14249"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://github.com/OSU-NLP-Group/ScienceAgentBench"], "content_sha256": "3974c4057ed46cb7a0964a0f90609f0cd2948d0993eab840196ef3096eda3ccd", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-243-scienceagentbench-toward-rigorous-assessment-of.raw.txt", "primary_url": "https://github.com/OSU-NLP-Group/ScienceAgentBench"}, "exemplar_id": "awesome_evals::ae-243-scienceagentbench-toward-rigorous-assessment-of", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-243-scienceagentbench-toward-rigorous-assessment-of::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-243-scienceagentbench-toward-rigorous-assessment-of", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "ScienceAgentBench: Toward Rigorous Assessment of Language Agents for Data-Driven Scientific Discovery", "source_type": "repository_or_docs", "source_url": "https://github.com/OSU-NLP-Group/ScienceAgentBench"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2409.11363"], "content_sha256": "63c7083a53e80039b9b9bfbb8acab6e170de506cbd510a311f738febcc2ec886", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-244-core-bench-computational-reproducibility-agent-b.raw.txt", "primary_url": "https://arxiv.org/abs/2409.11363"}, "exemplar_id": "awesome_evals::ae-244-core-bench-computational-reproducibility-agent-b", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-244-core-bench-computational-reproducibility-agent-b::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-244-core-bench-computational-reproducibility-agent-b", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "CORE-Bench: Computational Reproducibility Agent Benchmark", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2409.11363"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2506.11763"], "content_sha256": "0ec8046ae9b2178518200940143561109dbd49776c063dbb8e0da6b4a483318d", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/dream-deep-research-evaluation-agentic-metrics.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-245-deepresearch-bench-a-comprehensive-benchmark-for.raw.txt", "primary_url": "https://arxiv.org/abs/2506.11763"}, "exemplar_id": "awesome_evals::ae-245-deepresearch-bench-a-comprehensive-benchmark-for", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-245-deepresearch-bench-a-comprehensive-benchmark-for::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-245-deepresearch-bench-a-comprehensive-benchmark-for", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "DeepResearch Bench: A Comprehensive Benchmark for Deep Research Agents", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2506.11763"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2503.00096"], "content_sha256": "512d0cfac2e3e1516cf34e1d3b6783d3477d17b0c12f2a4e80c5e128d8bdf687", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/survey-evaluation-llm-based-agents-yehudai.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-246-bixbench-a-comprehensive-benchmark-for-llm-based.raw.txt", "primary_url": "https://arxiv.org/abs/2503.00096"}, "exemplar_id": "awesome_evals::ae-246-bixbench-a-comprehensive-benchmark-for-llm-based", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-246-bixbench-a-comprehensive-benchmark-for-llm-based::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-246-bixbench-a-comprehensive-benchmark-for-llm-based", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "BixBench: A Comprehensive Benchmark for LLM-based Agents in Computational Biology", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2503.00096"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2509.17158"], "content_sha256": "470306b50571e4c7017f191ab033468c939f1970d742655b4f2b29f181de7040", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/agentrewardbench-evaluating-automatic-evaluations-web-agent-.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-247-gaia2-and-are-scaling-up-agent-environments-and.raw.txt", "primary_url": "https://arxiv.org/abs/2509.17158"}, "exemplar_id": "awesome_evals::ae-247-gaia2-and-are-scaling-up-agent-environments-and", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-247-gaia2-and-are-scaling-up-agent-environments-and::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-247-gaia2-and-are-scaling-up-agent-environments-and", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "Gaia2 and ARE: Scaling Up Agent Environments and Evaluations", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2509.17158"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2502.15840"], "content_sha256": "fa3ae9bde739b9fafb9794e66029551c16e34ffe76c7b5e12cbc635715de3738", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/beyond-pass-at-1-reliability-science-long-horizon-agents.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-248-vending-bench-a-benchmark-for-long-term-coherenc.raw.txt", "primary_url": "https://arxiv.org/abs/2502.15840"}, "exemplar_id": "awesome_evals::ae-248-vending-bench-a-benchmark-for-long-term-coherenc", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-248-vending-bench-a-benchmark-for-long-term-coherenc::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-248-vending-bench-a-benchmark-for-long-term-coherenc", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2502.15840"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2505.11831"], "content_sha256": "d132a83ee12fc93c0fd1580cd389f0805f213e84d0e865221933b7e56603373f", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-249-arc-agi-2-a-new-challenge-for-frontier-ai-reason.raw.txt", "primary_url": "https://arxiv.org/abs/2505.11831"}, "exemplar_id": "awesome_evals::ae-249-arc-agi-2-a-new-challenge-for-frontier-ai-reason", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-249-arc-agi-2-a-new-challenge-for-frontier-ai-reason::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-249-arc-agi-2-a-new-challenge-for-frontier-ai-reason", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "ARC-AGI-2: A New Challenge for Frontier AI Reasoning Systems", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2505.11831"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2505.08638"], "content_sha256": "e41d0e6a6167b1e0c04afc559ffe871edc1a6c2a9c0d10317847409f77be74c9", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-250-trail-trace-reasoning-and-agentic-issue-localiza.raw.txt", "primary_url": "https://arxiv.org/abs/2505.08638"}, "exemplar_id": "awesome_evals::ae-250-trail-trace-reasoning-and-agentic-issue-localiza", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-250-trail-trace-reasoning-and-agentic-issue-localiza::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-250-trail-trace-reasoning-and-agentic-issue-localiza", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "TRAIL: Trace Reasoning and Agentic Issue Localization", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2505.08638"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://arxiv.org/abs/2505.18878"], "content_sha256": "e912a455f0144c6373c6956839924e07fbc34dca3826d57890f7702585e9e352", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-251-crmarena-pro-holistic-assessment-of-llm-agents-a.raw.txt", "primary_url": "https://arxiv.org/abs/2505.18878"}, "exemplar_id": "awesome_evals::ae-251-crmarena-pro-holistic-assessment-of-llm-agents-a", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-251-crmarena-pro-holistic-assessment-of-llm-agents-a::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-251-crmarena-pro-holistic-assessment-of-llm-agents-a", "source_section": "9 · Agent-specific evaluation (trajectories, tool use, multi-turn, world state, multi-agent, localization)", "source_title": "CRMArena-Pro: Holistic Assessment of LLM Agents Across Diverse Business Scenarios", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2505.18878"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://arxiv.org/abs/2605.12673"], "content_sha256": "4028581fc8ae51e495e4d40d23b30d19d020cfc73abb4daf62ceaeafb1698115", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/reliability-gap-agent-benchmarks-enterprise-simmering.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-252-benchjack-systematically-auditing-ai-agent-bench.raw.txt", "primary_url": "https://arxiv.org/abs/2605.12673"}, "exemplar_id": "awesome_evals::ae-252-benchjack-systematically-auditing-ai-agent-bench", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-252-benchjack-systematically-auditing-ai-agent-bench::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-252-benchjack-systematically-auditing-ai-agent-bench", "source_section": "10 · Safety / adversarial evaluation (prompt injection, jailbreaks, action-authorization, benchmark auditing)", "source_title": "BenchJack: Systematically Auditing AI Agent Benchmarks", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2605.12673"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://rdi.berkeley.edu/adv-llm-agents/slides/dawn-agentic-ai.pdf"], "content_sha256": "395a0add9ee6328bc4206a042b90dbc70063fc6a8aa0420ad4483ffe0d2cab6b", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-song-safe-secure-agentic.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-253-towards-building-safe-secure-agentic-ai.raw.txt", "primary_url": "https://rdi.berkeley.edu/adv-llm-agents/slides/dawn-agentic-ai.pdf"}, "exemplar_id": "awesome_evals::ae-253-towards-building-safe-secure-agentic-ai", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-253-towards-building-safe-secure-agentic-ai::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-253-towards-building-safe-secure-agentic-ai", "source_section": "10 · Safety / adversarial evaluation (prompt injection, jailbreaks, action-authorization, benchmark auditing)", "source_title": "Towards Building Safe & Secure Agentic AI", "source_type": "web_article", "source_url": "https://rdi.berkeley.edu/adv-llm-agents/slides/dawn-agentic-ai.pdf"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://iclr.cc/virtual/2025/invited-talk/36783"], "content_sha256": "89a19eff5d7204580467b2753ae672dd311b7515ff45111a648668ee02db2835", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-254-dawn-song-iclr-2025-keynote-on-llm-safety.raw.txt", "primary_url": "https://iclr.cc/virtual/2025/invited-talk/36783"}, "exemplar_id": "awesome_evals::ae-254-dawn-song-iclr-2025-keynote-on-llm-safety", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-254-dawn-song-iclr-2025-keynote-on-llm-safety::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-254-dawn-song-iclr-2025-keynote-on-llm-safety", "source_section": "10 · Safety / adversarial evaluation (prompt injection, jailbreaks, action-authorization, benchmark auditing)", "source_title": "Dawn Song — ICLR 2025 keynote on LLM safety", "source_type": "web_article", "source_url": "https://iclr.cc/virtual/2025/invited-talk/36783"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://arxiv.org/html/2506.02548v2"], "content_sha256": "b52a6f617753b95a185257b23542ced39ec7a2b9108625b5931ce2179d8e04ed", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-255-cybergym.raw.txt", "primary_url": "https://arxiv.org/html/2506.02548v2"}, "exemplar_id": "awesome_evals::ae-255-cybergym", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-255-cybergym::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-255-cybergym", "source_section": "10 · Safety / adversarial evaluation (prompt injection, jailbreaks, action-authorization, benchmark auditing)", "source_title": "CyberGym", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/html/2506.02548v2"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://arxiv.org/abs/2407.17436v2", "https://github.com/stanford-crfm/air-bench-2024"], "content_sha256": "bfeccc6ddd041c9a7cd63e42fe4afdd58f9d041385e3ff769ccc43d66bffaec6", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-256-air-bench-2024.raw.txt", "primary_url": "https://arxiv.org/abs/2407.17436v2"}, "exemplar_id": "awesome_evals::ae-256-air-bench-2024", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-256-air-bench-2024::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-256-air-bench-2024", "source_section": "10 · Safety / adversarial evaluation (prompt injection, jailbreaks, action-authorization, benchmark auditing)", "source_title": "AIR-Bench 2024", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2407.17436v2"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://decodingtrust.github.io"], "content_sha256": "ab3758cd0fd211cfd47ecd5beca43fd01b42084bea613f639cd717902fa8bc10", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-257-decodingtrust.raw.txt", "primary_url": "https://decodingtrust.github.io"}, "exemplar_id": "awesome_evals::ae-257-decodingtrust", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-257-decodingtrust::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-257-decodingtrust", "source_section": "10 · Safety / adversarial evaluation (prompt injection, jailbreaks, action-authorization, benchmark auditing)", "source_title": "DecodingTrust", "source_type": "web_article", "source_url": "https://decodingtrust.github.io"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://arxiv.org/abs/2411.07781"], "content_sha256": "8e4ca416e8d29a8a52d084f186996e38064ccc447d61deac305662dd7f262625", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-258-redcode.raw.txt", "primary_url": "https://arxiv.org/abs/2411.07781"}, "exemplar_id": "awesome_evals::ae-258-redcode", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-258-redcode::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-258-redcode", "source_section": "10 · Safety / adversarial evaluation (prompt injection, jailbreaks, action-authorization, benchmark auditing)", "source_title": "RedCode", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2411.07781"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://arxiv.org/abs/2407.12784"], "content_sha256": "ec75859c0dfe749e54eb9d62a48a2518cc3a2a5cd6ab17d97146cc053cb55c71", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-259-agentpoison.raw.txt", "primary_url": "https://arxiv.org/abs/2407.12784"}, "exemplar_id": "awesome_evals::ae-259-agentpoison", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-259-agentpoison::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-259-agentpoison", "source_section": "10 · Safety / adversarial evaluation (prompt injection, jailbreaks, action-authorization, benchmark auditing)", "source_title": "AgentPoison", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2407.12784"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://arxiv.org/abs/2411.00640", "https://www.anthropic.com/research/statistical-approach-to-model-evals"], "content_sha256": "1a3054a5cbcde63a165430ddf32e0f31c96a783b98b261ff523d81feb9d42711", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/paul-iusztin-ai-evals-dataset-error-analysis.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-260-adding-error-bars-to-evals-a-statistical-approac.raw.txt", "primary_url": "https://arxiv.org/abs/2411.00640"}, "exemplar_id": "awesome_evals::ae-260-adding-error-bars-to-evals-a-statistical-approac", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-260-adding-error-bars-to-evals-a-statistical-approac::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-260-adding-error-bars-to-evals-a-statistical-approac", "source_section": "10 · Safety / adversarial evaluation (prompt injection, jailbreaks, action-authorization, benchmark auditing)", "source_title": "Adding Error Bars to Evals (A Statistical Approach to LM Evaluations)", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2411.00640"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://arxiv.org/abs/2406.13352"], "content_sha256": "28b49b68f7f9500ed49d987dbb0db7e3e2ddc907157c2990d4d0b2a037dccd08", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-261-agentdojo-a-dynamic-environment-to-evaluate-prom.raw.txt", "primary_url": "https://arxiv.org/abs/2406.13352"}, "exemplar_id": "awesome_evals::ae-261-agentdojo-a-dynamic-environment-to-evaluate-prom", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-261-agentdojo-a-dynamic-environment-to-evaluate-prom::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-261-agentdojo-a-dynamic-environment-to-evaluate-prom", "source_section": "10 · Safety / adversarial evaluation (prompt injection, jailbreaks, action-authorization, benchmark auditing)", "source_title": "AgentDojo: A Dynamic Environment to Evaluate Prompt Injection Attacks and Defenses for LLM Agents", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2406.13352"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://arxiv.org/abs/2410.09024"], "content_sha256": "d4a15783464d29d25f3e5d7b021d868fcc36eb601ef4291ae3a18870d3cc4324", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-262-agentharm-a-benchmark-for-measuring-harmfulness.raw.txt", "primary_url": "https://arxiv.org/abs/2410.09024"}, "exemplar_id": "awesome_evals::ae-262-agentharm-a-benchmark-for-measuring-harmfulness", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-262-agentharm-a-benchmark-for-measuring-harmfulness::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-262-agentharm-a-benchmark-for-measuring-harmfulness", "source_section": "10 · Safety / adversarial evaluation (prompt injection, jailbreaks, action-authorization, benchmark auditing)", "source_title": "AgentHarm: A Benchmark for Measuring Harmfulness of LLM Agents", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2410.09024"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://arxiv.org/abs/2403.02691"], "content_sha256": "4493644eb2e357a86a9b9e8cc1b5bcba133daf9eaefce404ef4e5567aa966290", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-263-injecagent-benchmarking-indirect-prompt-injectio.raw.txt", "primary_url": "https://arxiv.org/abs/2403.02691"}, "exemplar_id": "awesome_evals::ae-263-injecagent-benchmarking-indirect-prompt-injectio", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-263-injecagent-benchmarking-indirect-prompt-injectio::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-263-injecagent-benchmarking-indirect-prompt-injectio", "source_section": "10 · Safety / adversarial evaluation (prompt injection, jailbreaks, action-authorization, benchmark auditing)", "source_title": "InjecAgent: Benchmarking Indirect Prompt Injections in Tool-Integrated LLM Agents", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2403.02691"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://arxiv.org/abs/2503.18813"], "content_sha256": "cfffce6b1f23371d6c485fbcc35fd180004724472403adccdb68eec7f46d9c79", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-264-defeating-prompt-injections-by-design-camel.raw.txt", "primary_url": "https://arxiv.org/abs/2503.18813"}, "exemplar_id": "awesome_evals::ae-264-defeating-prompt-injections-by-design-camel", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-264-defeating-prompt-injections-by-design-camel::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-264-defeating-prompt-injections-by-design-camel", "source_section": "10 · Safety / adversarial evaluation (prompt injection, jailbreaks, action-authorization, benchmark auditing)", "source_title": "Defeating Prompt Injections by Design (CaMeL)", "source_type": "paper_or_pdf", "source_url": "https://arxiv.org/abs/2503.18813"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://simonwillison.net/2025/Jun/16/the-lethal-trifecta/"], "content_sha256": "2dca7dc9d6b0a6a2f04daa46172fe3a4a2778bd9962276137bc2a6b4365c44f4", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-265-the-lethal-trifecta-for-ai-agents-private-data-u.raw.txt", "primary_url": "https://simonwillison.net/2025/Jun/16/the-lethal-trifecta/"}, "exemplar_id": "awesome_evals::ae-265-the-lethal-trifecta-for-ai-agents-private-data-u", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-265-the-lethal-trifecta-for-ai-agents-private-data-u::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-265-the-lethal-trifecta-for-ai-agents-private-data-u", "source_section": "10 · Safety / adversarial evaluation (prompt injection, jailbreaks, action-authorization, benchmark auditing)", "source_title": "The lethal trifecta for AI agents: private data, untrusted content, and external communication", "source_type": "web_article", "source_url": "https://simonwillison.net/2025/Jun/16/the-lethal-trifecta/"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://www.anthropic.com/research/shade-arena-sabotage-monitoring"], "content_sha256": "4e8c28f0f2b876c1a594e1ff0272240fd81348c3f662bc90b0935a1f6d5fa665", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/aws-evaluating-ai-agents-amazon.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-266-shade-arena-evaluating-sabotage-and-monitoring-i.raw.txt", "primary_url": "https://www.anthropic.com/research/shade-arena-sabotage-monitoring"}, "exemplar_id": "awesome_evals::ae-266-shade-arena-evaluating-sabotage-and-monitoring-i", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-266-shade-arena-evaluating-sabotage-and-monitoring-i::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-266-shade-arena-evaluating-sabotage-and-monitoring-i", "source_section": "10 · Safety / adversarial evaluation (prompt injection, jailbreaks, action-authorization, benchmark auditing)", "source_title": "SHADE-Arena: Evaluating Sabotage and Monitoring in LLM Agents", "source_type": "web_article", "source_url": "https://www.anthropic.com/research/shade-arena-sabotage-monitoring"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://www.anthropic.com/research/agentic-misalignment"], "content_sha256": "95259531ff732e93b05dc3febace62cd43cdb02d0dd6eca7cfb6e69e4163085b", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-267-agentic-misalignment-how-llms-could-be-insider-t.raw.txt", "primary_url": "https://www.anthropic.com/research/agentic-misalignment"}, "exemplar_id": "awesome_evals::ae-267-agentic-misalignment-how-llms-could-be-insider-t", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-267-agentic-misalignment-how-llms-could-be-insider-t::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-267-agentic-misalignment-how-llms-could-be-insider-t", "source_section": "10 · Safety / adversarial evaluation (prompt injection, jailbreaks, action-authorization, benchmark auditing)", "source_title": "Agentic Misalignment: How LLMs Could Be Insider Threats", "source_type": "web_article", "source_url": "https://www.anthropic.com/research/agentic-misalignment"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://github.com/Azure/PyRIT"], "content_sha256": "ed32f3ed26f566a29d0e9ed6077d5d8cdddc47c57202e994592ebd5766c986a6", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-268-pyrit-python-risk-identification-tool-for-genera.raw.txt", "primary_url": "https://github.com/Azure/PyRIT"}, "exemplar_id": "awesome_evals::ae-268-pyrit-python-risk-identification-tool-for-genera", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-268-pyrit-python-risk-identification-tool-for-genera::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-268-pyrit-python-risk-identification-tool-for-genera", "source_section": "10 · Safety / adversarial evaluation (prompt injection, jailbreaks, action-authorization, benchmark auditing)", "source_title": "PyRIT — Python Risk Identification Tool for generative AI", "source_type": "repository_or_docs", "source_url": "https://github.com/Azure/PyRIT"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://genai.owasp.org/resource/owasp-top-10-for-agentic-applications-for-2026/"], "content_sha256": "79608e9dc39f72e762bb95c504b1893a0fad91dde0fb3e1120c3bcfa0e83042f", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-269-owasp-top-10-for-agentic-applications-2026-llm-a.raw.txt", "primary_url": "https://genai.owasp.org/resource/owasp-top-10-for-agentic-applications-for-2026/"}, "exemplar_id": "awesome_evals::ae-269-owasp-top-10-for-agentic-applications-2026-llm-a", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-269-owasp-top-10-for-agentic-applications-2026-llm-a::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-269-owasp-top-10-for-agentic-applications-2026-llm-a", "source_section": "10 · Safety / adversarial evaluation (prompt injection, jailbreaks, action-authorization, benchmark auditing)", "source_title": "OWASP Top 10 for Agentic Applications (2026) + LLM Applications (2025)", "source_type": "web_article", "source_url": "https://genai.owasp.org/resource/owasp-top-10-for-agentic-applications-for-2026/"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://atlas.mitre.org/"], "content_sha256": "a332bcff1db7814b2fba13da36f924c610de89da03c7eb29593be8b527bdd8e6", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/papers/adversarial-examples-evaluating-reading-comprehension-systems-ji.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-270-mitre-atlas-adversarial-threat-landscape-for-ai.raw.txt", "primary_url": "https://atlas.mitre.org/"}, "exemplar_id": "awesome_evals::ae-270-mitre-atlas-adversarial-threat-landscape-for-ai", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-270-mitre-atlas-adversarial-threat-landscape-for-ai::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-270-mitre-atlas-adversarial-threat-landscape-for-ai", "source_section": "10 · Safety / adversarial evaluation (prompt injection, jailbreaks, action-authorization, benchmark auditing)", "source_title": "MITRE ATLAS — Adversarial Threat Landscape for AI Systems", "source_type": "web_article", "source_url": "https://atlas.mitre.org/"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://proceedings.iclr.cc/paper_files/paper/2025/file/5750f91d8fb9d5c02bd8ad2c3b44456b-Paper-Conference.pdf"], "content_sha256": "e2505f8632bfcb6a64a4390a3170b3ca1dfd3f9916d7c3cf9ba2b89887b3a0c9", "http_status": 200, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/ankur-goyal-agent-driven-benchmarking-evals.md", "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-271-agent-security-bench-asb-formalizing-and-benchma.raw.txt", "primary_url": "https://proceedings.iclr.cc/paper_files/paper/2025/file/5750f91d8fb9d5c02bd8ad2c3b44456b-Paper-Conference.pdf"}, "exemplar_id": "awesome_evals::ae-271-agent-security-bench-asb-formalizing-and-benchma", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-271-agent-security-bench-asb-formalizing-and-benchma::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-271-agent-security-bench-asb-formalizing-and-benchma", "source_section": "10 · Safety / adversarial evaluation (prompt injection, jailbreaks, action-authorization, benchmark auditing)", "source_title": "Agent Security Bench (ASB): Formalizing and Benchmarking Attacks and Defenses in LLM-based Agents", "source_type": "paper_or_pdf", "source_url": "https://proceedings.iclr.cc/paper_files/paper/2025/file/5750f91d8fb9d5c02bd8ad2c3b44456b-Paper-Conference.pdf"}
{"eval_domain": "benchmark_integrity", "evidence": {"all_urls": ["https://app.grayswan.ai/arena/blog/agent-red-teaming-the-ai-jailbreak-showdown"], "content_sha256": null, "http_status": 429, "local_note_path": null, "local_raw_path": null, "primary_url": "https://app.grayswan.ai/arena/blog/agent-red-teaming-the-ai-jailbreak-showdown"}, "exemplar_id": "awesome_evals::ae-272-gray-swan-x-uk-aisi-agent-red-teaming-challenge", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-272-gray-swan-x-uk-aisi-agent-red-teaming-challenge::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "benchmark_validity_hazard", "source_id": "ae-272-gray-swan-x-uk-aisi-agent-red-teaming-challenge", "source_section": "10 · Safety / adversarial evaluation (prompt injection, jailbreaks, action-authorization, benchmark auditing)", "source_title": "Gray Swan x UK AISI Agent Red-Teaming Challenge", "source_type": "blog", "source_url": "https://app.grayswan.ai/arena/blog/agent-red-teaming-the-ai-jailbreak-showdown"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=eLXF0VojuSs"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-hamel-sedgh-domain-eval-systems.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=eLXF0VojuSs"}, "exemplar_id": "awesome_evals::ae-273-how-to-construct-domain-specific-llm-evaluation", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-273-how-to-construct-domain-specific-llm-evaluation::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-273-how-to-construct-domain-specific-llm-evaluation", "source_section": "🎤 Conference & individual talks", "source_title": "How to Construct Domain Specific LLM Evaluation Systems", "source_type": "talk_video", "source_url": "https://www.youtube.com/watch?v=eLXF0VojuSs"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=jryZvCuA0Uc"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-huber-liu-look-at-your-data.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=jryZvCuA0Uc"}, "exemplar_id": "awesome_evals::ae-274-how-to-look-at-your-data", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-274-how-to-look-at-your-data::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-274-how-to-look-at-your-data", "source_section": "🎤 Conference & individual talks", "source_title": "How to look at your data", "source_type": "talk_video", "source_url": "https://www.youtube.com/watch?v=jryZvCuA0Uc"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=k98gDjYbSaU"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-bischof-failure-is-a-funnel.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=k98gDjYbSaU"}, "exemplar_id": "awesome_evals::ae-275-failure-is-a-funnel", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-275-failure-is-a-funnel::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-275-failure-is-a-funnel", "source_section": "🎤 Conference & individual talks", "source_title": "Failure is a Funnel", "source_type": "talk_video", "source_url": "https://www.youtube.com/watch?v=k98gDjYbSaU"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=7EGF0Mc0_os"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-yan-llms-as-judges.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=7EGF0Mc0_os"}, "exemplar_id": "awesome_evals::ae-276-using-llms-as-judges-insights-challenges-best-pr", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-276-using-llms-as-judges-insights-challenges-best-pr::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-276-using-llms-as-judges-insights-challenges-best-pr", "source_section": "🎤 Conference & individual talks", "source_title": "Using LLMs as Judges: Insights, Challenges, Best Practices", "source_type": "talk_video", "source_url": "https://www.youtube.com/watch?v=7EGF0Mc0_os"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=eGVDKegRdgM"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-shankar-scaling-vibe-checks.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=eGVDKegRdgM"}, "exemplar_id": "awesome_evals::ae-277-scaling-up-vibe-checks-for-llms", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-277-scaling-up-vibe-checks-for-llms::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-277-scaling-up-vibe-checks-for-llms", "source_section": "🎤 Conference & individual talks", "source_title": "Scaling Up Vibe Checks for LLMs", "source_type": "talk_video", "source_url": "https://www.youtube.com/watch?v=eGVDKegRdgM"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=H-1QaLPnGsg"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-shankar-why-pipelines-fail.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=H-1QaLPnGsg"}, "exemplar_id": "awesome_evals::ae-278-why-llm-data-processing-pipelines-fail", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-278-why-llm-data-processing-pipelines-fail::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-278-why-llm-data-processing-pipelines-fail", "source_section": "🎤 Conference & individual talks", "source_title": "Why LLM Data Processing Pipelines Fail", "source_type": "talk_video", "source_url": "https://www.youtube.com/watch?v=H-1QaLPnGsg"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=L8OoYeDI_ls"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-pesok-evals-not-unit-tests.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=L8OoYeDI_ls"}, "exemplar_id": "awesome_evals::ae-279-evals-are-not-unit-tests", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-279-evals-are-not-unit-tests::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-279-evals-are-not-unit-tests", "source_section": "🎤 Conference & individual talks", "source_title": "Evals Are Not Unit Tests", "source_type": "talk_video", "source_url": "https://www.youtube.com/watch?v=L8OoYeDI_ls"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=jxrGodnopHo"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-karam-metrics-that-work.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=jxrGodnopHo"}, "exemplar_id": "awesome_evals::ae-280-building-metrics-that-actually-work-workshop", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-280-building-metrics-that-actually-work-workshop::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-280-building-metrics-that-actually-work-workshop", "source_section": "🎤 Conference & individual talks", "source_title": "Building Metrics that actually work (workshop)", "source_type": "talk_video", "source_url": "https://www.youtube.com/watch?v=jxrGodnopHo"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=kDczF4wBh8s"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-hopkins-self-driving-voice-agents.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=kDczF4wBh8s"}, "exemplar_id": "awesome_evals::ae-281-from-self-driving-to-autonomous-voice-agents", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-281-from-self-driving-to-autonomous-voice-agents::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-281-from-self-driving-to-autonomous-voice-agents", "source_section": "🎤 Conference & individual talks", "source_title": "From Self-driving to Autonomous Voice Agents", "source_type": "talk_video", "source_url": "https://www.youtube.com/watch?v=kDczF4wBh8s"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=OMGPvW8TBHc"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-tang-fuzzing-genai.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=OMGPvW8TBHc"}, "exemplar_id": "awesome_evals::ae-282-fuzzing-in-the-genai-era", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-282-fuzzing-in-the-genai-era::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-282-fuzzing-in-the-genai-era", "source_section": "🎤 Conference & individual talks", "source_title": "Fuzzing in the GenAI Era", "source_type": "talk_video", "source_url": "https://www.youtube.com/watch?v=OMGPvW8TBHc"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=qdmxApz3EJI"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-khattab-systems-that-endure.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=qdmxApz3EJI"}, "exemplar_id": "awesome_evals::ae-283-on-engineering-ai-systems-that-endure-the-bitter", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-283-on-engineering-ai-systems-that-endure-the-bitter::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-283-on-engineering-ai-systems-that-endure-the-bitter", "source_section": "🎤 Conference & individual talks", "source_title": "On Engineering AI Systems that Endure the Bitter Lesson", "source_type": "talk_video", "source_url": "https://www.youtube.com/watch?v=qdmxApz3EJI"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=89NuzmKokIk"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-smith-strategies-for-llm-evals.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=89NuzmKokIk"}, "exemplar_id": "awesome_evals::ae-284-strategies-for-llm-evals-harnesses-workshop", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-284-strategies-for-llm-evals-harnesses-workshop::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-284-strategies-for-llm-evals-harnesses-workshop", "source_section": "🎤 Conference & individual talks", "source_title": "Strategies for LLM Evals (harnesses workshop)", "source_type": "talk_video", "source_url": "https://www.youtube.com/watch?v=89NuzmKokIk"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=MC55hdWLq4o"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-goyal-future-of-evals.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=MC55hdWLq4o"}, "exemplar_id": "awesome_evals::ae-285-the-future-of-evals", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-285-the-future-of-evals::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-285-the-future-of-evals", "source_section": "🎤 Conference & individual talks", "source_title": "The Future of Evals", "source_type": "talk_video", "source_url": "https://www.youtube.com/watch?v=MC55hdWLq4o"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=b6Doq2fz81U"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-wei-3-key-ideas-2025.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=b6Doq2fz81U"}, "exemplar_id": "awesome_evals::ae-286-3-key-ideas-in-ai-in-2025-verifier-s-law", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-286-3-key-ideas-in-ai-in-2025-verifier-s-law::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-286-3-key-ideas-in-ai-in-2025-verifier-s-law", "source_section": "🎤 Conference & individual talks", "source_title": "3 Key Ideas in AI in 2025 (Verifier's Law)", "source_type": "talk_video", "source_url": "https://www.youtube.com/watch?v=b6Doq2fz81U"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=l898fqkjdFc"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/papers/evaluating-large-language-models-trained-on-code.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=l898fqkjdFc"}, "exemplar_id": "awesome_evals::ae-287-some-intuitions-about-large-language-models", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-287-some-intuitions-about-large-language-models::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-287-some-intuitions-about-large-language-models", "source_section": "🎤 Conference & individual talks", "source_title": "Some Intuitions About Large Language Models", "source_type": "talk_video", "source_url": "https://www.youtube.com/watch?v=l898fqkjdFc"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=7xTGNNLPyMI"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-karpathy-deep-dive-llms.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=7xTGNNLPyMI"}, "exemplar_id": "awesome_evals::ae-288-deep-dive-into-llms-like-chatgpt", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-288-deep-dive-into-llms-like-chatgpt::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-288-deep-dive-into-llms-like-chatgpt", "source_section": "🎤 Conference & individual talks", "source_title": "Deep Dive into LLMs like ChatGPT", "source_type": "talk_video", "source_url": "https://www.youtube.com/watch?v=7xTGNNLPyMI"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=hhiLw5Q_UFg"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-schulman-rlhf-progress-challenges.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=hhiLw5Q_UFg"}, "exemplar_id": "awesome_evals::ae-289-rlhf-progress-and-challenges", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-289-rlhf-progress-and-challenges::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-289-rlhf-progress-and-challenges", "source_section": "🎤 Conference & individual talks", "source_title": "RLHF: Progress and Challenges", "source_type": "talk_video", "source_url": "https://www.youtube.com/watch?v=hhiLw5Q_UFg"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=AdLgPmcrXwQ"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/papers/evaluating-large-language-models-trained-on-code.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=AdLgPmcrXwQ"}, "exemplar_id": "awesome_evals::ae-290-aligning-open-language-models", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-290-aligning-open-language-models::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-290-aligning-open-language-models", "source_section": "🎤 Conference & individual talks", "source_title": "Aligning Open Language Models", "source_type": "talk_video", "source_url": "https://www.youtube.com/watch?v=AdLgPmcrXwQ"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=spamOhG7BOA"], "content_sha256": null, "http_status": null, "local_note_path": null, "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=spamOhG7BOA"}, "exemplar_id": "awesome_evals::ae-291-building-llm-applications-for-production", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-291-building-llm-applications-for-production::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-291-building-llm-applications-for-production", "source_section": "🎤 Conference & individual talks", "source_title": "Building LLM Applications for Production", "source_type": "talk_video", "source_url": "https://www.youtube.com/watch?v=spamOhG7BOA"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=4dUFIRj-BWo"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-lee-model-is-the-product.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=4dUFIRj-BWo"}, "exemplar_id": "awesome_evals::ae-292-the-model-is-the-product", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-292-the-model-is-the-product::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-292-the-model-is-the-product", "source_section": "🎤 Conference & individual talks", "source_title": "The Model is the Product", "source_type": "talk_video", "source_url": "https://www.youtube.com/watch?v=4dUFIRj-BWo"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=_IzZWeuTx7I"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-brown-rl-environments-at-scale.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=_IzZWeuTx7I"}, "exemplar_id": "awesome_evals::ae-293-rl-environments-at-scale", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-293-rl-environments-at-scale::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-293-rl-environments-at-scale", "source_section": "🎤 Conference & individual talks", "source_title": "RL Environments at Scale", "source_type": "talk_video", "source_url": "https://www.youtube.com/watch?v=_IzZWeuTx7I"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=kmTMc-fVSXw"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-brand-benchmarks-time-of-agents.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=kmTMc-fVSXw"}, "exemplar_id": "awesome_evals::ae-294-llm-benchmarks-in-the-time-of-agents", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-294-llm-benchmarks-in-the-time-of-agents::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-294-llm-benchmarks-in-the-time-of-agents", "source_section": "🎤 Conference & individual talks", "source_title": "LLM benchmarks in the time of agents", "source_type": "talk_video", "source_url": "https://www.youtube.com/watch?v=kmTMc-fVSXw"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=PgzOBNse2EA"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/paul-iusztin-ai-evals-dataset-error-analysis.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=PgzOBNse2EA"}, "exemplar_id": "awesome_evals::ae-295-evals-error-analysis-and-better-prompts", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-295-evals-error-analysis-and-better-prompts::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-295-evals-error-analysis-and-better-prompts", "source_section": "🎙 Podcast episodes", "source_title": "Evals, error analysis, and better prompts", "source_type": "podcast_video", "source_url": "https://www.youtube.com/watch?v=PgzOBNse2EA"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=QE_1hRLsehM"], "content_sha256": null, "http_status": null, "local_note_path": null, "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=QE_1hRLsehM"}, "exemplar_id": "awesome_evals::ae-296-evals-are-the-new-prd-for-ai-products", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-296-evals-are-the-new-prd-for-ai-products::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-296-evals-are-the-new-prd-for-ai-products", "source_section": "🎙 Podcast episodes", "source_title": "Evals are the new PRD for AI products", "source_type": "podcast_video", "source_url": "https://www.youtube.com/watch?v=QE_1hRLsehM"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=QEk-XwrkqhI"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-pod-vg60-10-things-i-hate.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=QEk-XwrkqhI"}, "exemplar_id": "awesome_evals::ae-297-ep-60-10-things-i-hate-about-ai-evals", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-297-ep-60-10-things-i-hate-about-ai-evals::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-297-ep-60-10-things-i-hate-about-ai-evals", "source_section": "🎙 Podcast episodes", "source_title": "Ep 60: 10 Things I Hate About AI Evals", "source_type": "podcast_video", "source_url": "https://www.youtube.com/watch?v=QEk-XwrkqhI"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=rWToRi2_SeY"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-pod-vg50-field-guide.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=rWToRi2_SeY"}, "exemplar_id": "awesome_evals::ae-298-ep-50-a-field-guide-to-rapidly-improving-ai-prod", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-298-ep-50-a-field-guide-to-rapidly-improving-ai-prod::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-298-ep-50-a-field-guide-to-rapidly-improving-ai-prod", "source_section": "🎙 Podcast episodes", "source_title": "Ep 50: A Field Guide to Rapidly Improving AI Products", "source_type": "podcast_video", "source_url": "https://www.youtube.com/watch?v=rWToRi2_SeY"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=a4BV0gGmXgA"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-pod-ls-goyal-five-lessons.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=a4BV0gGmXgA"}, "exemplar_id": "awesome_evals::ae-299-five-hard-earned-lessons-about-evals", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-299-five-hard-earned-lessons-about-evals::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-299-five-hard-earned-lessons-about-evals", "source_section": "🎙 Podcast episodes", "source_title": "Five Hard-Earned Lessons About Evals", "source_type": "podcast_video", "source_url": "https://www.youtube.com/watch?v=a4BV0gGmXgA"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=v5mBjeX4TJ8"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/artificial-analysis-independent-llm-evals-latent-space.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=v5mBjeX4TJ8"}, "exemplar_id": "awesome_evals::ae-300-artificial-analysis-independent-llm-evals", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-300-artificial-analysis-independent-llm-evals::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-300-artificial-analysis-independent-llm-evals", "source_section": "🎙 Podcast episodes", "source_title": "Artificial Analysis: Independent LLM Evals", "source_type": "podcast_video", "source_url": "https://www.youtube.com/watch?v=v5mBjeX4TJ8"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=ZAimcoJXUBo"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/articles/andon-labs-reality-final-eval-latent-space.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=ZAimcoJXUBo"}, "exemplar_id": "awesome_evals::ae-301-reality-the-final-eval-vending-bench", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-301-reality-the-final-eval-vending-bench::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-301-reality-the-final-eval-vending-bench", "source_section": "🎙 Podcast episodes", "source_title": "Reality: The Final Eval (Vending-Bench)", "source_type": "podcast_video", "source_url": "https://www.youtube.com/watch?v=ZAimcoJXUBo"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=-N6MajRfqYw"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-pod-aitw5-designing-evals.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=-N6MajRfqYw"}, "exemplar_id": "awesome_evals::ae-302-5-designing-evals", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-302-5-designing-evals::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-302-5-designing-evals", "source_section": "🎙 Podcast episodes", "source_title": "#5 Designing Evals", "source_type": "podcast_video", "source_url": "https://www.youtube.com/watch?v=-N6MajRfqYw"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=OawyQOrlubM"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/papers/evaluating-large-language-models-trained-on-code.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=OawyQOrlubM"}, "exemplar_id": "awesome_evals::ae-303-16-evaluating-prompts-across-models", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-303-16-evaluating-prompts-across-models::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-303-16-evaluating-prompts-across-models", "source_section": "🎙 Podcast episodes", "source_title": "#16 Evaluating Prompts Across Models", "source_type": "podcast_video", "source_url": "https://www.youtube.com/watch?v=OawyQOrlubM"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=5Fy0hBzyduU"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-pod-aitw24-classification-evals.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=5Fy0hBzyduU"}, "exemplar_id": "awesome_evals::ae-304-24-evals-for-classification", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-304-24-evals-for-classification::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-304-24-evals-for-classification", "source_section": "🎙 Podcast episodes", "source_title": "#24 Evals for Classification", "source_type": "podcast_video", "source_url": "https://www.youtube.com/watch?v=5Fy0hBzyduU"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=jzhVo0iAX_I"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-pod-aitw34-multimodal-evals.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=jzhVo0iAX_I"}, "exemplar_id": "awesome_evals::ae-305-34-multimodal-evals", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-305-34-multimodal-evals::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-305-34-multimodal-evals", "source_section": "🎙 Podcast episodes", "source_title": "#34 Multimodal Evals", "source_type": "podcast_video", "source_url": "https://www.youtube.com/watch?v=jzhVo0iAX_I"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=9EjWR3QpJYk"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-pod-mlops372-still-talking-evals.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=9EjWR3QpJYk"}, "exemplar_id": "awesome_evals::ae-306-372-it-s-2026-and-we-re-still-talking-evals", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-306-372-it-s-2026-and-we-re-still-talking-evals::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-306-372-it-s-2026-and-we-re-still-talking-evals", "source_section": "🎙 Podcast episodes", "source_title": "#372 It's 2026 and We're Still Talking Evals", "source_type": "podcast_video", "source_url": "https://www.youtube.com/watch?v=9EjWR3QpJYk"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=3kbiGPn0cOo"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-pod-twiml728-generative-benchmarking.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=3kbiGPn0cOo"}, "exemplar_id": "awesome_evals::ae-307-728-generative-benchmarking", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-307-728-generative-benchmarking::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-307-728-generative-benchmarking", "source_section": "🎙 Podcast episodes", "source_title": "#728 Generative Benchmarking", "source_type": "podcast_video", "source_url": "https://www.youtube.com/watch?v=3kbiGPn0cOo"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=kwkdKirqi6s"], "content_sha256": null, "http_status": null, "local_note_path": null, "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=kwkdKirqi6s"}, "exemplar_id": "awesome_evals::ae-308-shaping-ai-benchmarks-helm", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-308-shaping-ai-benchmarks-helm::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-308-shaping-ai-benchmarks-helm", "source_section": "🎙 Podcast episodes", "source_title": "Shaping AI Benchmarks (HELM)", "source_type": "podcast_video", "source_url": "https://www.youtube.com/watch?v=kwkdKirqi6s"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=okHMaczHPXc"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/papers/bertscore-evaluating-text-generation-with-bert.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=okHMaczHPXc"}, "exemplar_id": "awesome_evals::ae-309-evaluating-llms-with-chatbot-arena", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-309-evaluating-llms-with-chatbot-arena::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-309-evaluating-llms-with-chatbot-arena", "source_section": "🎙 Podcast episodes", "source_title": "Evaluating LLMs with Chatbot Arena", "source_type": "podcast_video", "source_url": "https://www.youtube.com/watch?v=okHMaczHPXc"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=v0eTTn7ZPEc"], "content_sha256": null, "http_status": null, "local_note_path": null, "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=v0eTTn7ZPEc"}, "exemplar_id": "awesome_evals::ae-310-evaluating-ai-designing-for-non-determinism", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-310-evaluating-ai-designing-for-non-determinism::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-310-evaluating-ai-designing-for-non-determinism", "source_section": "🎙 Podcast episodes", "source_title": "Evaluating AI, Designing for Non-Determinism", "source_type": "podcast_video", "source_url": "https://www.youtube.com/watch?v=v0eTTn7ZPEc"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=-lRBpyPt79c"], "content_sha256": null, "http_status": null, "local_note_path": null, "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=-lRBpyPt79c"}, "exemplar_id": "awesome_evals::ae-311-karpathy-rl-is-terrible-why-benchmarks-mislead", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-311-karpathy-rl-is-terrible-why-benchmarks-mislead::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-311-karpathy-rl-is-terrible-why-benchmarks-mislead", "source_section": "🎙 Podcast episodes", "source_title": "Karpathy: RL is terrible, why benchmarks mislead", "source_type": "podcast_video", "source_url": "https://www.youtube.com/watch?v=-lRBpyPt79c"}
{"eval_domain": "talks_podcasts_slides", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=J7N9FMouSKg"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-talk-hamel-shreya-build-evals-2026.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=J7N9FMouSKg"}, "exemplar_id": "awesome_evals::ae-312-how-to-build-ai-evals-in-2026-step-by-step", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-312-how-to-build-ai-evals-in-2026-step-by-step::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unstructured_practitioner_claim", "source_id": "ae-312-how-to-build-ai-evals-in-2026-step-by-step", "source_section": "🎙 Podcast episodes", "source_title": "How to Build AI Evals in 2026 (Step-by-Step)", "source_type": "podcast_video", "source_url": "https://www.youtube.com/watch?v=J7N9FMouSKg"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=QAgR4uQ15rc"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-song-safe-trustworthy-agents.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=QAgR4uQ15rc"}, "exemplar_id": "awesome_evals::ae-313-towards-building-safe-trustworthy-ai-agents", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-313-towards-building-safe-trustworthy-ai-agents::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-313-towards-building-safe-trustworthy-ai-agents", "source_section": "🎓 University lectures", "source_title": "Towards Building Safe & Trustworthy AI Agents", "source_type": "lecture_video", "source_url": "https://www.youtube.com/watch?v=QAgR4uQ15rc"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=ti6yPE2VPZc"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-song-safe-secure-agentic.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=ti6yPE2VPZc"}, "exemplar_id": "awesome_evals::ae-314-towards-building-safe-and-secure-agentic-ai", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-314-towards-building-safe-and-secure-agentic-ai::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-314-towards-building-safe-and-secure-agentic-ai", "source_section": "🎓 University lectures", "source_title": "Towards Building Safe and Secure Agentic AI", "source_type": "lecture_video", "source_url": "https://www.youtube.com/watch?v=ti6yPE2VPZc"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=6y2AnWol7oo"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-mann-measuring-capabilities-rsp.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=6y2AnWol7oo"}, "exemplar_id": "awesome_evals::ae-315-measuring-agent-capabilities-and-anthropic-s-rsp", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-315-measuring-agent-capabilities-and-anthropic-s-rsp::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-315-measuring-agent-capabilities-and-anthropic-s-rsp", "source_section": "🎓 University lectures", "source_title": "Measuring Agent Capabilities and Anthropic's RSP", "source_type": "lecture_video", "source_url": "https://www.youtube.com/watch?v=6y2AnWol7oo"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=f3KKx9LWntQ"], "content_sha256": null, "http_status": null, "local_note_path": null, "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=f3KKx9LWntQ"}, "exemplar_id": "awesome_evals::ae-316-open-source-and-science-in-the-era-of-foundation", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-316-open-source-and-science-in-the-era-of-foundation::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-316-open-source-and-science-in-the-era-of-foundation", "source_section": "🎓 University lectures", "source_title": "Open-Source and Science in the Era of Foundation Models", "source_type": "lecture_video", "source_url": "https://www.youtube.com/watch?v=f3KKx9LWntQ"}
{"eval_domain": "general_eval_source", "evidence": {"all_urls": ["https://www.youtube.com/watch?v=x-R5l2HsXqM"], "content_sha256": null, "http_status": null, "local_note_path": "sources/21-benchmarks/awesome-evals-primary-sources/notes/talks/talk-cs336-lec12-evaluation.md", "local_raw_path": null, "primary_url": "https://www.youtube.com/watch?v=x-R5l2HsXqM"}, "exemplar_id": "awesome_evals::ae-317-cs336-lecture-12-evaluation", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "transcript", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-317-cs336-lecture-12-evaluation::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unclassified_eval_failure", "source_id": "ae-317-cs336-lecture-12-evaluation", "source_section": "🎓 University lectures", "source_title": "CS336 Lecture 12: Evaluation", "source_type": "lecture_video", "source_url": "https://www.youtube.com/watch?v=x-R5l2HsXqM"}
{"eval_domain": "verifier_or_rl_environment", "evidence": {"all_urls": ["https://pavlovslist.com/"], "content_sha256": "b0a232e118083b74166f8d4a92270f2d48c8ecbbfb23937973e92dbc56fec67a", "http_status": 200, "local_note_path": null, "local_raw_path": "sources/21-benchmarks/awesome-evals-primary-sources/raw/ae-318-pavlovslist-com.raw.txt", "primary_url": "https://pavlovslist.com/"}, "exemplar_id": "awesome_evals::ae-318-pavlovslist-com", "golden_case": {"acceptance_checks": ["failure is classified where it occurs in the agentic cycle", "raw source evidence is retained outside the normalized row", "public benchmark claims are not treated as private enterprise evidence", "judge, harness, model, and tool versions are recorded when applicable"], "expected_artifact_type": "source_document", "expected_fields": ["harness.name", "harness.version", "model.provider", "model.name", "runtime_context.prompt_template_id", "runtime_context.retrieval_index_id", "attached_tools", "trace_ref", "failure_code", "source_bundle_ref"], "input": "Classify a failed AI-agent run using this source's eval concept and preserve the evidence boundary.", "persona": "enterprise_eval_owner", "task_id": "ae-318-pavlovslist-com::classify_failure"}, "quality_bar": "golden_quality_candidate", "recommended_failure_code": "unverifiable_reward", "source_id": "ae-318-pavlovslist-com", "source_section": "Companies & landscape (eval / RL-environment market)", "source_title": "pavlovslist.com", "source_type": "web_article", "source_url": "https://pavlovslist.com/"}
