{"schema_version":"eval_source_gap_v1","gap_id":"gap_deepeval_local_test_run_captured_20260622","framework":"DeepEval","source_ledger_ref":"sources/21-benchmarks/eval-native-schema-boundaries-2026.json#DeepEval.remaining_gaps[0]","gap_text":"Package-pinned local no-key TestRun JSON is captured for deepeval==4.0.6","status":"captured","gap_type":"captured_evidence","verification_needed":"Keep native_deepeval_test_run_real_run_001 validating with raw export hash.","owner_role":"eval-platform","priority":"p3","related_fixture_ids":["native_deepeval_test_run_real_run_001"],"related_raw_exports":["statebench/fixtures/enterprise-eval-contract/v1/deepeval-test-run-realrun-4.0.6.json"],"notes":"Captured evidence row retained so the ledger can mix open gaps with verified package fixtures without ambiguity.","created_at_utc":"2026-06-22T12:25:00Z"}
{"schema_version":"eval_source_gap_v1","gap_id":"gap_deepeval_confident_ai_export_20260622","framework":"DeepEval","source_ledger_ref":"sources/21-benchmarks/eval-native-schema-boundaries-2026.json#DeepEval.remaining_gaps[1]","gap_text":"Hosted Confident AI export shape still needs capture","status":"open","gap_type":"hosted_export","verification_needed":"Run or obtain a Confident AI hosted project export from a pinned DeepEval/Confident AI version and store raw payload plus hash.","owner_role":"eval-platform","priority":"p1","related_fixture_ids":[],"related_raw_exports":[],"notes":"Keep separate from local TestRun JSON because hosted project exports can include dashboard-only fields.","created_at_utc":"2026-06-22T12:25:00Z"}
{"schema_version":"eval_source_gap_v1","gap_id":"gap_deepeval_component_trace_export_20260622","framework":"DeepEval","source_ledger_ref":"sources/21-benchmarks/eval-native-schema-boundaries-2026.json#DeepEval.remaining_gaps[2]","gap_text":"Component trace export shape still needs capture","status":"open","gap_type":"trace_export","verification_needed":"Capture a component trace export with retrieval/tool/evaluator spans and map raw span fields to trace_ref_v1.","owner_role":"eval-platform","priority":"p1","related_fixture_ids":[],"related_raw_exports":[],"notes":"Needed before claiming DeepEval coverage for agentic chat-cycle failures.","created_at_utc":"2026-06-22T12:25:00Z"}
{"schema_version":"eval_source_gap_v1","gap_id":"gap_deepeval_builtin_llm_judge_20260622","framework":"DeepEval","source_ledger_ref":"sources/21-benchmarks/eval-native-schema-boundaries-2026.json#DeepEval.remaining_gaps[3]","gap_text":"Built-in LLM judge metrics still need pinned fixtures with model/provider metadata","status":"open","gap_type":"package_fixture","verification_needed":"Run a built-in DeepEval LLM metric with explicit judge model/provider metadata and store the local TestRun JSON.","owner_role":"eval-platform","priority":"p1","related_fixture_ids":[],"related_raw_exports":[],"notes":"The current fixture uses a deterministic custom BaseMetric and intentionally avoids API keys.","created_at_utc":"2026-06-22T12:25:00Z"}
{"schema_version":"eval_source_gap_v1","gap_id":"gap_evaluate_save_captured_20260622","framework":"Hugging Face Evaluate","source_ledger_ref":"sources/21-benchmarks/eval-native-schema-boundaries-2026.json#Hugging Face Evaluate.remaining_gaps[0]","gap_text":"Package-pinned evaluate.save serialization is captured for evaluate==0.4.6","status":"captured","gap_type":"captured_evidence","verification_needed":"Keep native_huggingface_evaluate_save_real_run_001 validating with raw export hash.","owner_role":"eval-platform","priority":"p3","related_fixture_ids":["native_huggingface_evaluate_save_real_run_001"],"related_raw_exports":["statebench/fixtures/enterprise-eval-contract/v1/evaluate-realrun-0.4.6.json"],"notes":"Captured serialization only; metric execution remains separate.","created_at_utc":"2026-06-22T12:25:00Z"}
{"schema_version":"eval_source_gap_v1","gap_id":"gap_evaluate_metric_module_execution_20260622","framework":"Hugging Face Evaluate","source_ledger_ref":"sources/21-benchmarks/eval-native-schema-boundaries-2026.json#Hugging Face Evaluate.remaining_gaps[1]","gap_text":"evaluate.load metric-module execution still needs selected per-metric pinned fixtures","status":"open","gap_type":"package_fixture","verification_needed":"Run selected evaluate.load metrics offline or with pinned Hub revisions and store output dictionaries plus runtime metadata.","owner_role":"ml-platform","priority":"p2","related_fixture_ids":[],"related_raw_exports":[],"notes":"Needed for per-metric adapters; Evaluate intentionally lacks one universal output schema.","created_at_utc":"2026-06-22T12:25:00Z"}
{"schema_version":"eval_source_gap_v1","gap_id":"gap_evaluate_suite_cache_artifacts_20260622","framework":"Hugging Face Evaluate","source_ledger_ref":"sources/21-benchmarks/eval-native-schema-boundaries-2026.json#Hugging Face Evaluate.remaining_gaps[2]","gap_text":"EvaluationSuite and distributed cache artifacts still need pinned captures","status":"open","gap_type":"package_fixture","verification_needed":"Run EvaluationSuite and distributed/cache paths with pinned versions and store emitted artifacts.","owner_role":"ml-platform","priority":"p2","related_fixture_ids":[],"related_raw_exports":[],"notes":"Relevant for model benchmark reproducibility and shared runners.","created_at_utc":"2026-06-22T12:25:00Z"}
{"schema_version":"eval_source_gap_v1","gap_id":"gap_ragas_dataset_jsonl_captured_20260622","framework":"RAGAS","source_ledger_ref":"sources/21-benchmarks/eval-native-schema-boundaries-2026.json#RAGAS.remaining_gaps[0]","gap_text":"Package-pinned dataset JSONL serialization is captured for ragas==0.4.3","status":"captured","gap_type":"captured_evidence","verification_needed":"Keep native_ragas_dataset_jsonl_real_run_001 validating with raw export hash.","owner_role":"eval-platform","priority":"p3","related_fixture_ids":["native_ragas_dataset_jsonl_real_run_001"],"related_raw_exports":["statebench/fixtures/enterprise-eval-contract/v1/ragas-dataset-realrun-0.4.3.jsonl"],"notes":"Covers sample serialization, not scored results.","created_at_utc":"2026-06-22T12:25:00Z"}
{"schema_version":"eval_source_gap_v1","gap_id":"gap_ragas_scored_evaluation_result_20260622","framework":"RAGAS","source_ledger_ref":"sources/21-benchmarks/eval-native-schema-boundaries-2026.json#RAGAS.remaining_gaps[1]","gap_text":"Scored EvaluationResult dataframe/JSON shape still needs a pinned run with valid ragas_traces","status":"open","gap_type":"package_fixture","verification_needed":"Run RAGAS scoring with a deterministic or pinned judge/embedding setup and capture EvaluationResult, dataframe/JSON, cost, traces, and ragas_traces fields.","owner_role":"eval-platform","priority":"p1","related_fixture_ids":[],"related_raw_exports":[],"notes":"Required before treating RAGAS metric rows as production adapter evidence.","created_at_utc":"2026-06-22T12:25:00Z"}
{"schema_version":"eval_source_gap_v1","gap_id":"gap_ragas_metric_required_fields_20260622","framework":"RAGAS","source_ledger_ref":"sources/21-benchmarks/eval-native-schema-boundaries-2026.json#RAGAS.remaining_gaps[2]","gap_text":"Metric-specific required-field matrix still needs capture for selected enterprise RAG metrics","status":"open","gap_type":"metric_mapping","verification_needed":"For selected RAGAS metrics, record required sample fields, optional fields, failure behavior, and normalization mapping.","owner_role":"eval-platform","priority":"p1","related_fixture_ids":[],"related_raw_exports":[],"notes":"Needed to prevent adapters from running metrics against incomplete sample rows.","created_at_utc":"2026-06-22T12:25:00Z"}
{"schema_version":"eval_source_gap_v1","gap_id":"gap_promptfoo_no_json_schema_20260622","framework":"Promptfoo","source_ledger_ref":"sources/21-benchmarks/eval-native-schema-boundaries-2026.json#Promptfoo.remaining_gaps[0]","gap_text":"No standalone published JSON Schema for exports","status":"open","gap_type":"schema_absence","verification_needed":"Continue pinning Promptfoo package versions and validating raw JSON/JSONL exports because no standalone upstream JSON Schema is available.","owner_role":"eval-platform","priority":"p2","related_fixture_ids":["native_promptfoo_json_export_real_run_001","native_promptfoo_jsonl_export_real_run_001"],"related_raw_exports":["statebench/fixtures/enterprise-eval-contract/v1/promptfoo-realrun-0.121.17.json","statebench/fixtures/enterprise-eval-contract/v1/promptfoo-realrun-0.121.17.jsonl"],"notes":"This is a permanent adapter posture unless Promptfoo publishes stable schemas.","created_at_utc":"2026-06-22T12:25:00Z"}
{"schema_version":"eval_source_gap_v1","gap_id":"gap_promptfoo_redteam_report_export_20260622","framework":"Promptfoo","source_ledger_ref":"sources/21-benchmarks/eval-native-schema-boundaries-2026.json#Promptfoo.remaining_gaps[1]","gap_text":"Separate red-team report UI/export schema needs deeper source tracing","status":"open","gap_type":"hosted_export","verification_needed":"Capture Promptfoo red-team report exports or UI data payloads and map plugin/strategy dimensions to failure taxonomy.","owner_role":"security-eval","priority":"p1","related_fixture_ids":[],"related_raw_exports":[],"notes":"Normal eval JSON/JSONL is pinned; red-team reporting remains separate.","created_at_utc":"2026-06-22T12:25:00Z"}
{"schema_version":"eval_source_gap_v1","gap_id":"gap_promptfoo_provider_variants_20260622","framework":"Promptfoo","source_ledger_ref":"sources/21-benchmarks/eval-native-schema-boundaries-2026.json#Promptfoo.remaining_gaps[2]","gap_text":"Version-pinned fixture needed before production adapter lock","status":"captured","gap_type":"captured_evidence","verification_needed":"Promptfoo 0.121.17 JSON and JSONL echo-provider fixtures are captured; add provider-specific variants before locking provider adapters.","owner_role":"eval-platform","priority":"p2","related_fixture_ids":["native_promptfoo_json_export_real_run_001","native_promptfoo_jsonl_export_real_run_001"],"related_raw_exports":["statebench/fixtures/enterprise-eval-contract/v1/promptfoo-realrun-0.121.17.json","statebench/fixtures/enterprise-eval-contract/v1/promptfoo-realrun-0.121.17.jsonl"],"notes":"Captured for generic export shape, still limited for provider-specific response bodies.","created_at_utc":"2026-06-22T12:25:00Z"}
{"schema_version":"eval_source_gap_v1","gap_id":"gap_openevals_typescript_types_20260622","framework":"OpenEvals","source_ledger_ref":"sources/21-benchmarks/eval-native-schema-boundaries-2026.json#OpenEvals.remaining_gaps[0]","gap_text":"TypeScript package object types need the same source-level pass","status":"open","gap_type":"package_fixture","verification_needed":"Inspect OpenEvals TypeScript source and capture evaluator result shapes and custom-output behavior.","owner_role":"eval-platform","priority":"p2","related_fixture_ids":[],"related_raw_exports":[],"notes":"Python object shape is captured; TS package may differ.","created_at_utc":"2026-06-22T12:25:00Z"}
{"schema_version":"eval_source_gap_v1","gap_id":"gap_openevals_langsmith_feedback_payload_20260622","framework":"OpenEvals","source_ledger_ref":"sources/21-benchmarks/eval-native-schema-boundaries-2026.json#OpenEvals.remaining_gaps[1]","gap_text":"LangSmith docs do not expose a full serialized feedback/run payload for OpenEvals","status":"open","gap_type":"hosted_export","verification_needed":"Capture OpenEvals-to-LangSmith feedback serialization from a real run or official export.","owner_role":"eval-platform","priority":"p2","related_fixture_ids":[],"related_raw_exports":[],"notes":"Needed when OpenEvals is used as a LangSmith feedback writer rather than a standalone result object.","created_at_utc":"2026-06-22T12:25:00Z"}
{"schema_version":"eval_source_gap_v1","gap_id":"gap_evaluate_no_universal_schema_20260622","framework":"Hugging Face Evaluate","source_ledger_ref":"sources/21-benchmarks/eval-native-schema-boundaries-2026.json#Hugging Face Evaluate.remaining_gaps[0]","gap_text":"Upstream intentionally has no universal metric result schema","status":"open","gap_type":"schema_absence","verification_needed":"Treat each selected metric as a separate adapter contract with fixture-backed key mapping.","owner_role":"ml-platform","priority":"p2","related_fixture_ids":["native_huggingface_evaluate_save_real_run_001"],"related_raw_exports":["statebench/fixtures/enterprise-eval-contract/v1/evaluate-realrun-0.4.6.json"],"notes":"This duplicate framework section covers metric-boundary posture, not evaluate.save serialization.","created_at_utc":"2026-06-22T12:25:00Z"}
{"schema_version":"eval_source_gap_v1","gap_id":"gap_evaluate_enterprise_metric_mappings_20260622","framework":"Hugging Face Evaluate","source_ledger_ref":"sources/21-benchmarks/eval-native-schema-boundaries-2026.json#Hugging Face Evaluate.remaining_gaps[1]","gap_text":"Enterprise adapters need per-metric mappings for selected metrics","status":"open","gap_type":"metric_mapping","verification_needed":"Declare metric-specific mappings for accuracy, exact match, F1, toxicity/safety, and domain-specific metrics selected for enterprise use.","owner_role":"ml-platform","priority":"p2","related_fixture_ids":[],"related_raw_exports":[],"notes":"Avoids treating arbitrary Evaluate metric dictionaries as normalized eval_result_v1 rows.","created_at_utc":"2026-06-22T12:25:00Z"}
{"schema_version":"eval_source_gap_v1","gap_id":"gap_lighteval_full_cli_run_20260622","framework":"Hugging Face LightEval","source_ledger_ref":"sources/21-benchmarks/eval-native-schema-boundaries-2026.json#Hugging Face LightEval.remaining_gaps[0]","gap_text":"Need full LightEval CLI run over a real model/dataset to confirm task config population and backend-specific export variations","status":"open","gap_type":"package_fixture","verification_needed":"Run LightEval CLI with a pinned tiny model/dataset or controlled local task and capture result JSON, detail parquet, config_tasks, and backend metadata.","owner_role":"ml-platform","priority":"p1","related_fixture_ids":["native_lighteval_logger_bundle_real_run_001"],"related_raw_exports":["statebench/fixtures/enterprise-eval-contract/v1/lighteval-logger-realrun-0.13.0.json"],"notes":"Logger-level fixture is useful but does not exercise full task loading or model inference.","created_at_utc":"2026-06-22T12:25:00Z"}
{"schema_version":"eval_source_gap_v1","gap_id":"gap_lighteval_logger_fixture_captured_20260622","framework":"Hugging Face LightEval","source_ledger_ref":"sources/21-benchmarks/eval-native-schema-boundaries-2026.json#Hugging Face LightEval.remaining_gaps[1]","gap_text":"Logger-level lighteval==0.13.0 fixture confirms result JSON sections and detail parquet columns doc, metric, and model_response","status":"captured","gap_type":"captured_evidence","verification_needed":"Keep native_lighteval_logger_bundle_real_run_001 validating with raw export hash.","owner_role":"ml-platform","priority":"p3","related_fixture_ids":["native_lighteval_logger_bundle_real_run_001"],"related_raw_exports":["statebench/fixtures/enterprise-eval-contract/v1/lighteval-logger-realrun-0.13.0.json"],"notes":"Captured by EvaluationTracker.save, not by CLI.","created_at_utc":"2026-06-22T12:25:00Z"}
{"schema_version":"eval_source_gap_v1","gap_id":"gap_openai_zdr_hosted_tracing_limitation_20260622","framework":"OpenAI Agents SDK","source_ledger_ref":"sources/21-benchmarks/eval-native-schema-boundaries-2026.json#OpenAI Agents SDK.remaining_gaps[0]","gap_text":"Hosted tracing unavailable for OpenAI API ZDR organizations","status":"open","gap_type":"platform_limitation","verification_needed":"Document custom trace processor requirements and capture raw processor export fixtures for ZDR-compatible deployments.","owner_role":"agent-platform","priority":"p1","related_fixture_ids":[],"related_raw_exports":[],"notes":"Enterprise adapters cannot assume hosted trace visibility.","created_at_utc":"2026-06-22T12:25:00Z"}
{"schema_version":"eval_source_gap_v1","gap_id":"gap_openai_custom_trace_processor_fixture_20260622","framework":"OpenAI Agents SDK","source_ledger_ref":"sources/21-benchmarks/eval-native-schema-boundaries-2026.json#OpenAI Agents SDK.remaining_gaps[1]","gap_text":"Production adapter should capture raw processor/export payloads from a concrete trace processor","status":"open","gap_type":"trace_export","verification_needed":"Run Agents SDK with a custom trace processor and store raw generation/tool/guardrail/handoff span payloads.","owner_role":"agent-platform","priority":"p1","related_fixture_ids":[],"related_raw_exports":[],"notes":"Needed for trace_ref_v1 adapter proof.","created_at_utc":"2026-06-22T12:25:00Z"}
{"schema_version":"eval_source_gap_v1","gap_id":"gap_google_agent_engine_event_export_20260622","framework":"Google ADK","source_ledger_ref":"sources/21-benchmarks/eval-native-schema-boundaries-2026.json#Google ADK.remaining_gaps[0]","gap_text":"Google Agent Engine platform event export shape still needs capture beyond open-source ADK","status":"open","gap_type":"provider_export","verification_needed":"Capture Google Agent Engine event/export payloads and compare against open-source ADK Event and EvalSet objects.","owner_role":"agent-platform","priority":"p1","related_fixture_ids":[],"related_raw_exports":[],"notes":"Open-source ADK fields are captured; hosted platform exports may differ.","created_at_utc":"2026-06-22T12:25:00Z"}
{"schema_version":"eval_source_gap_v1","gap_id":"gap_microsoft_foundry_eval_exports_20260622","framework":"Microsoft AutoGen","source_ledger_ref":"sources/21-benchmarks/eval-native-schema-boundaries-2026.json#Microsoft AutoGen.remaining_gaps[0]","gap_text":"Microsoft Foundry-specific eval and telemetry exports were not verified in this pass","status":"open","gap_type":"provider_export","verification_needed":"Capture Foundry eval and telemetry export payloads and map any AutoGen/OpenTelemetry overlap.","owner_role":"agent-platform","priority":"p2","related_fixture_ids":[],"related_raw_exports":[],"notes":"AutoGen OTel runtime source is covered; Foundry platform export remains separate.","created_at_utc":"2026-06-22T12:25:00Z"}
{"schema_version":"eval_source_gap_v1","gap_id":"gap_mistral_observability_judge_openapi_20260622","framework":"Mistral Agents and Conversations","source_ledger_ref":"sources/21-benchmarks/eval-native-schema-boundaries-2026.json#Mistral Agents and Conversations.remaining_gaps[0]","gap_text":"Mistral observability judge/eval object schemas need OpenAPI-level extraction","status":"open","gap_type":"provider_export","verification_needed":"Extract Mistral OpenAPI objects for observability, judge, and eval payloads and add source-observed fixtures.","owner_role":"agent-platform","priority":"p2","related_fixture_ids":[],"related_raw_exports":[],"notes":"Conversation output entries are covered; eval/judge objects are not.","created_at_utc":"2026-06-22T12:25:00Z"}
{"schema_version":"eval_source_gap_v1","gap_id":"gap_qwen_agent_trace_envelope_20260622","framework":"Qwen-Agent","source_ledger_ref":"sources/21-benchmarks/eval-native-schema-boundaries-2026.json#Qwen-Agent.remaining_gaps[0]","gap_text":"No first-class Qwen-Agent trace/event telemetry envelope found in primary repo files","status":"open","gap_type":"schema_absence","verification_needed":"Confirm whether Qwen-Agent has a durable trace/event envelope or define a wrapper adapter around message and ReAct stream records.","owner_role":"agent-platform","priority":"p2","related_fixture_ids":[],"related_raw_exports":[],"notes":"Message/function-call stream is covered; first-class telemetry was not found.","created_at_utc":"2026-06-22T12:25:00Z"}
