{
  "purpose": "Source recovery only; text-extracted URLs may be truncated by line wrapping. Not live-verified, endorsed, or claim-linked.",
  "sources": [
    {
      "source_filename": "Best GitHub Repositories for LLM Evaluation in 2026.pdf",
      "source_sha256": "68758a8bd2a6ad3540150c11568b9d8505f184b8116f13bf2ddf3082cb972f09",
      "extraction": "pdftotext",
      "urls": [
        "https://github.com/Arize-ai/phoenix",
        "https://github.com/EleutherAI/lm-evaluation-harness.git",
        "https://github.com/EleutherAI/lm-evaluation-harness/commits/main/",
        "https://github.com/EleutherAI/lm-evaluation-harness?ref=Technology",
        "https://github.com/EvolvingLMMs-Lab/lmms-eval",
        "https://github.com/Giskard-AI/giskard",
        "https://github.com/HumanSignal/label-studio",
        "https://github.com/SWE-bench/SWE-bench",
        "https://github.com/UKGovernmentBEIS/inspect_ai",
        "https://github.com/argilla-io/argilla",
        "https://github.com/braintrustdata/autoevals",
        "https://github.com/comet-ml/opik",
        "https://github.com/comet-ml/opik/commits/main/",
        "https://github.com/confident-ai/deepeval",
        "https://github.com/explodinggradients/ragas",
        "https://github.com/google/BIG-bench",
        "https://github.com/huggingface/lighteval",
        "https://github.com/langfuse/langfuse",
        "https://github.com/langfuse/langfuse/commits/main/",
        "https://github.com/open-compass/opencompass",
        "https://github.com/openai/evals",
        "https://github.com/openai/simple-evals?utm_source=chatgpt.com",
        "https://github.com/promptfoo/promptfoo",
        "https://github.com/promptfoo/promptfoo/commits/main/",
        "https://github.com/truera/trulens"
      ]
    },
    {
      "source_filename": "Deep Research Report_ Best Courses and YouTube Resources for Learning LLM Evaluation.pdf",
      "source_sha256": "104c887e876195029ee1eb387ff9ffeb89135d7eead90e9e6ca7906df60ab52d",
      "extraction": "pdftotext",
      "urls": [
        "https://academy.langchain.com/courses/intro-to-langsmith",
        "https://arize.com/ai-courses-and-certifications/",
        "https://arxiv.org/abs/2303.16634?utm_source=chatgpt.com",
        "https://arxiv.org/abs/2310.08491?utm_source=chatgpt.com",
        "https://arxiv.org/html/2606.13685v1?utm_source=chatgpt.com",
        "https://developers.openai.com/api/docs/guides/evals",
        "https://github.com/Arize-ai/phoenix",
        "https://github.com/EleutherAI/lm-evaluation-harness/",
        "https://github.com/confident-ai/deepeval",
        "https://github.com/huggingface/lighteval",
        "https://github.com/promptfoo/promptfoo",
        "https://github.com/stanford-crfm/helm",
        "https://huggingface.co/learn/llm-course/en/chapter11/5",
        "https://maven.com/ai-evals-and-analytics/ai-evals-analytics-playbook",
        "https://maven.com/p/175365",
        "https://maven.com/parlance-labs/evals",
        "https://www.deeplearning.ai/courses/evaluating-debugging-generative-ai?utm_source=chatgpt.com",
        "https://www.evidentlyai.com/llm-evaluation-course-practice",
        "https://www.youtube.com/watch?v=9iN-cPnp7xg&utm_source=chatgpt.com",
        "https://www.youtube.com/watch?v=Xfl50508LZM&utm_source=chatgpt.com",
        "https://www.youtube.com/watch?v=rHs0sP7b5fM&utm_source=chatgpt.com",
        "https://www.youtube.com/watch?v=uiza7wp1KrE&utm_source=chatgpt.com"
      ]
    },
    {
      "source_filename": "Eval-Driven AI Programming_ A Code-First Path to Production AI Engineering.docx",
      "source_sha256": "dba3959ffcc8f85f796bc032730d621e206442f95e7a72c68629d61c7d46e5f4",
      "extraction": "macOS textutil",
      "urls": [
        "https://arize.com/docs/phoenix",
        "https://arize.com/docs/phoenix/evaluation/pre-built-metrics/tool-selection",
        "https://arize.com/docs/phoenix/evaluation/server-evals/builtin-evaluators",
        "https://arxiv.org/abs/2404.12272",
        "https://arxiv.org/abs/2606.19057",
        "https://docs.langchain.com/langsmith/evaluate-complex-agent",
        "https://docs.langchain.com/langsmith/evaluation",
        "https://docs.langchain.com/langsmith/pytest",
        "https://docs.langchain.com/langsmith/trajectory-evals",
        "https://docs.wandb.ai/weave/guides/evaluation/scorers",
        "https://github.com/Arize-ai/phoenix/tree/main/tutorials/evals",
        "https://github.com/UKGovernmentBEIS/inspect_ai",
        "https://github.com/UKGovernmentBEIS/inspect_evals",
        "https://github.com/UKGovernmentBEIS/inspect_evals/blob/main/src/inspect_evals/swe_bench/README.md",
        "https://github.com/braintrustdata/braintrust-cookbook",
        "https://github.com/langchain-ai/agentevals",
        "https://github.com/openai/openai-cookbook.git",
        "https://github.com/openai/openai-cookbook/tree/main/examples/evaluation",
        "https://github.com/promptfoo/promptfoo-action",
        "https://github.com/promptfoo/promptfoo/blob/main/.claude/skills/promptfoo-evals/SKILL.md",
        "https://github.com/promptfoo/promptfoo/blob/main/examples/simple-test/promptfooconfig.yaml",
        "https://google.github.io/agents-cli/guide/evaluation/",
        "https://hamel.dev/blog/posts/eval-smell/",
        "https://hamel.dev/blog/posts/evals-faq/are-similarity-metrics-bertscore-rouge-etc-useful-for-evaluating-llm-outputs.html",
        "https://hamel.dev/blog/posts/evals-faq/how-often-should-i-re-run-error-analysis-on-my-production-system.html",
        "https://hamel.dev/blog/posts/evals-faq/should-i-practice-eval-driven-development.html",
        "https://humanloop.com/docs/changelog/2025/08",
        "https://maven.com/p/0c0359/improve-ai-consistently-getting-started-with-evals",
        "https://maven.com/p/a58f3f/how-to-setup-evals-for-agents",
        "https://maven.com/p/a945a7/setting-eval-for-ai-agents-scaling-with-auto-evaluation",
        "https://maven.com/p/d2dc30/how-open-ai-customers-use-evals-to-build-better-ai-products",
        "https://maven.com/parlance-labs/evals",
        "https://openai.com/index/openai-to-acquire-promptfoo/",
        "https://platform.openai.com/docs/api-reference/graders?api-mode=chat",
        "https://platform.openai.com/docs/assistants/deep-dive/run-lifecycle%23.webm",
        "https://platform.openai.com/docs/quickstart/make-your-first-api-request",
        "https://wandb.ai/site/courses/weave/",
        "https://www.anthropic.com/engineering/demystifying-evals-for-ai-agents",
        "https://www.anthropic.com/engineering/infrastructure-noise",
        "https://www.braintrust.dev/docs/evaluate",
        "https://www.youtube.com/watch?v=D7_ipDqhtwk"
      ]
    },
    {
      "source_filename": "LLM Evaluation Around OpenAI Deep Research_ First-Party Posts, Benchmarks, and the Deep-Tech Referen.pdf",
      "source_sha256": "462d1de6982626cecdeaf21620b695b31c317b80f051a54217319618f78fb94a",
      "extraction": "pdftotext",
      "urls": [
        "https://agi.safe.ai/",
        "https://agi.safe.ai/;",
        "https://agi.safe.ai/?utm_source=chatgpt.com",
        "https://arxiv.org/abs/",
        "https://arxiv.org/abs/2501.14249",
        "https://arxiv.org/abs/2501.14249?utm_source=chatgpt.com",
        "https://arxiv.org/abs/2504.12516",
        "https://cdn.openai.com/",
        "https://cdn.openai.com/pdf/5e10f4ab-d6f7-442e-9508-59515c65e35d/browsecomp.pdf",
        "https://deploymentsafety.openai.com/deep-research",
        "https://docs.gptr.dev/",
        "https://docs.gptr.dev/blog/2025/02/26/deep-research",
        "https://evals.futuresearch.ai/",
        "https://evals.futuresearch.ai/?utm_source=chatgpt.com",
        "https://futuresearch.ai/effort-paradox/?utm_source=chatgpt.com",
        "https://github.com/openai/",
        "https://github.com/openai/simple-evals",
        "https://github.com/openai/simpleevals",
        "https://huggingface.co/",
        "https://huggingface.co/blog/open-deep-research",
        "https://huggingface.co/datasets/",
        "https://huggingface.co/datasets/cais/hle?utm_source=chatgpt.com",
        "https://huggingface.co/datasets/gaia-benchmark/GAIA?utm_source=chatgpt.com",
        "https://huggingface.co/gaia-benchmark?utm_source=chatgpt.com",
        "https://huggingface.co/gaiabenchmark",
        "https://ii.inc/blog/post/ii-researcher",
        "https://ii.inc/blog/post/iiresearcher",
        "https://leehanchung.github.io/blogs/2025/02/26/deep-research/?utm_source=chatgpt.com",
        "https://openai.com/index/",
        "https://openai.com/index/browsecomp/",
        "https://openai.com/index/deep-research-system-card/",
        "https://openai.com/index/introducing-deep-research/?utm_source=chatgpt.com",
        "https://openai.com/pa-IN/index/introducing-deep-research/",
        "https://parallel.ai/blog/",
        "https://parallel.ai/blog/deep-research-benchmarks",
        "https://thezvi.substack.com/p/ai-105-hey-there-alexa?utm_source=chatgpt.com",
        "https://www.datacamp.com/blog/deep-research-openai",
        "https://www.interconnects.ai/p/deep-research-information-vs-insight-in-science",
        "https://www.oneusefulthing.org/p/the-end-of-search-the-beginning-of",
        "https://www.science.org/",
        "https://www.science.org/content/blog-post/evaluation-deep-research-performance?utm_source=chatgpt.com",
        "https://www.together.ai/",
        "https://www.together.ai/blog/open-deep-research",
        "https://www.together.ai/blog/open-deep-research?utm_source=chatgpt.com"
      ]
    },
    {
      "source_filename": "LLM and AI Evaluation Interview Question Bank_ Deep Research Report.pdf",
      "source_sha256": "4584cb0ecec960e5c97a6458a3460cbc8be2e56977bfc1c879179b37d4df2259",
      "extraction": "pdftotext",
      "urls": [
        "https://aclanthology.org/2020.acl-main.442/",
        "https://aclanthology.org/2022.findings-acl.165/",
        "https://aclanthology.org/2023.emnlp-main.153/?utm_source=chatgpt.com",
        "https://aclanthology.org/2024.eacl-demo.16/",
        "https://aclanthology.org/2026.acl-long.635/",
        "https://aclanthology.org/P02-1040/",
        "https://arxiv.org/abs/1706.04599",
        "https://arxiv.org/abs/2009.03300?utm_source=chatgpt.com",
        "https://arxiv.org/abs/2009.11462",
        "https://arxiv.org/abs/2107.03374",
        "https://arxiv.org/abs/2109.07958",
        "https://arxiv.org/abs/2110.14168",
        "https://arxiv.org/abs/2203.02155",
        "https://arxiv.org/abs/2206.04615?utm_source=chatgpt.com",
        "https://arxiv.org/abs/2210.09261?utm_source=chatgpt.com",
        "https://arxiv.org/abs/2211.09110?utm_source=chatgpt.com",
        "https://arxiv.org/abs/2305.14251",
        "https://arxiv.org/abs/2306.04528",
        "https://arxiv.org/abs/2306.05685",
        "https://arxiv.org/abs/2308.01263",
        "https://arxiv.org/abs/2308.03688",
        "https://arxiv.org/abs/2310.06770",
        "https://arxiv.org/abs/2311.12022?utm_source=chatgpt.com",
        "https://arxiv.org/abs/2311.12983",
        "https://arxiv.org/abs/2311.14648",
        "https://arxiv.org/abs/2402.04249",
        "https://arxiv.org/abs/2402.10260",
        "https://arxiv.org/abs/2403.04132",
        "https://arxiv.org/abs/2404.01318",
        "https://arxiv.org/abs/2502.18848",
        "https://arxiv.org/html/2405.14782v3",
        "https://arxiv.org/html/2406.04244v1",
        "https://developers.openai.com/api/docs/guides/evaluation-best-practices",
        "https://github.com/amitshekhariitbhu/ai-engineering-interview-questions",
        "https://github.com/confident-ai/deepeval",
        "https://github.com/eleutherai/lm-evaluation-harness",
        "https://github.com/promptfoo/promptfoo",
        "https://inspect.aisi.org.uk/",
        "https://mlflow.org/docs/latest/genai/eval-monitor/",
        "https://prachub.com/resources/llm-evaluation-interview-questions-evals-llm-as-judge-drift-and-production-quality",
        "https://prachub.com/resources/llm-evaluation-interview-questions-evals-llm-as-judge-drift-andproduction-quality",
        "https://www.nist.gov/itl/ai-risk-management-framework"
      ]
    },
    {
      "source_filename": "x Evals, Observability and Release Gates for Production AI Systems.docx",
      "source_sha256": "847a5176d41cf2c0600168c4fa1789480618efe0b2f8fc6e5dcbc7a288cf8491",
      "extraction": "macOS textutil",
      "urls": [
        "https://argo-rollouts.readthedocs.io/en/stable/features/analysis/",
        "https://arize.com/docs/phoenix",
        "https://arize.com/docs/phoenix/tracing/llm-traces/metrics",
        "https://arxiv.org/abs/2406.07791",
        "https://digital-strategy.ec.europa.eu/en/faqs/navigating-ai-act",
        "https://docs.aws.amazon.com/sagemaker/latest/dg/model-validation.html",
        "https://docs.databricks.com/gcp/en/mlflow3/genai/eval-monitor/production-monitoring",
        "https://docs.evidentlyai.com/introduction",
        "https://docs.feast.dev/v0.60-branch",
        "https://docs.langchain.com/langsmith/evaluation",
        "https://evals.openai.com/",
        "https://langfuse.com/docs",
        "https://mlflow.org/releases/3.14.0/",
        "https://openai.com/fr-FR/index/introducing-agentkit/",
        "https://openai.com/it-IT/index/evals-drive-next-chapter-of-ai/",
        "https://openai.com/tr-TR/index/inside-our-in-house-data-agent/",
        "https://opentelemetry.io/docs/",
        "https://opentelemetry.io/docs/specs/semconv/registry/attributes/gen-ai/",
        "https://sre.google/workbook/canarying-releases/",
        "https://wandb.ai/site/weave-new",
        "https://www.anthropic.com/engineering/a-postmortem-of-three-recent-issues",
        "https://www.anthropic.com/engineering/demystifying-evals-for-ai-agents",
        "https://www.braintrust.dev/docs",
        "https://www.nist.gov/publications/artificial-intelligence-risk-management-framework-generative-artificial-intelligence"
      ]
    }
  ]
}
