{
  "schema_version": "1.0.0",
  "evidence_release": "public-evidence-v1-2026-08-26",
  "scientific_scope": "PubMed titles and abstracts; full-text methods, controls, statistics, contradictions, and reproducibility were not assessed",
  "benchmark": {
    "id": "bioasq14b-disease-focus-18-v1",
    "case_count": 18,
    "question_types": {
      "yesno": 11,
      "summary": 5,
      "list": 1,
      "factoid": 1
    },
    "development_case_count": 12,
    "validation_case_count": 6,
    "limitations": [
      "BioASQ reference publications and answers are evaluator-only and never retrieval inputs",
      "BioASQ reference publications may not include every relevant paper, and ideal answers are comparison answers rather than verified support for every claim",
      "These cases previously informed BinfoNet development, so they measure development progress rather than performance on unseen questions; independent testing requires a locked set of new, expert-reviewed questions"
    ]
  },
  "retrieval": {
    "policy": "coverage_then_semantic",
    "development": {
      "case_count": 12,
      "macro_candidate_recall": 1.0,
      "macro_recall_at_10": 0.3394681677,
      "macro_ndcg_at_10": 0.4886388698,
      "macro_mrr_at_10": 0.5891203704
    },
    "validation": {
      "case_count": 6,
      "macro_candidate_recall": 0.9848484848,
      "macro_recall_at_10": 0.4755411255,
      "macro_ndcg_at_10": 0.4278993332,
      "macro_mrr_at_10": 0.5238095238,
      "candidate_generation_miss_count": 1,
      "zero_recall_at_10_case_count": 1
    },
    "interpretation": "Candidate recall measures whether reference publications entered the candidate pool. Recall@10, nDCG@10, and MRR@10 measure different properties of the ranked top ten. These metrics do not establish scientific evidence quality."
  },
  "synthesis": {
    "record_count": 36,
    "end_to_end_record_count": 18,
    "reference_evidence_record_count": 18,
    "schema_valid_count": 36,
    "provenance_valid_count": 36,
    "yesno_accuracy": {
      "end_to_end": 0.6363636364,
      "reference_evidence": 0.7272727273,
      "case_count_per_arm": 11
    },
    "status": "two_arm_synthesis_complete",
    "interpretation": "The arm difference is descriptive and does not establish that retrieval caused the answer difference. Semantic and ROUGE similarity are descriptive only, not factuality or groundedness measures."
  },
  "automated_judge": {
    "status": "provisional_awaiting_expert_calibration",
    "end_to_end_claim_count": 71,
    "end_to_end_supported_label_count": 71,
    "dominant_label_fraction": 1.0,
    "quality_flag": "low_label_diversity_requires_expert_and_mutation_calibration",
    "interpretation": "The automated judge result is a diagnostic output, not evidence that generated answers are scientifically correct."
  },
  "observability": {
    "phoenix_experiment_count": 4,
    "phoenix_task_trace_count": 36,
    "tracked_screenshot": "../../assets/evidence/phoenix-retrieval-experiments-v1.png"
  },
  "assessment": {
    "release_readiness": "not_assessed",
    "expert_review": "awaiting_genuine_review",
    "recommendation": "Continue evaluation before using outputs for biomedical decision support"
  },
  "source_artifact_sha256": {
    "synthesis_evaluation_summary": "518069d16dc4eefcce9ef4272165602f44699b69bed5f304961aab970159cb16",
    "retrieval_validation_summary": "984cc2b71c03dcce52a8fe72b5cd3a543c3b25a13103a9346a20ba4758596be9",
    "retrieval_development_summary": "c43ed54346fe71db506481fa38b67731157c92da7285d740ac3a003dcdb67d12",
    "phoenix_screenshot": "f013283541c031060fd79e323a4dd8938da1023865c860bf2b3160b623ef60ae"
  }
}
