{
  "ok": true,
  "clientSlug": "appen",
  "services": [
    {
      "name": "AI research services",
      "description": "AI research services are managed research operations that embed external domain expertise, evaluation infrastructure, and data pipelines directly into a model team's workflow. Appen provides these as an extension of your research program (context engineering, LLM-as-a-judge evaluation, synthetic data, and benchmarking) run to the quality standards your research reputation demands.",
      "pricingNote": "",
      "availability": "available",
      "url": "https://www.appen.com/ai-research-services"
    },
    {
      "name": "Custom benchmarking and evaluation",
      "description": "Bespoke benchmarks and evaluation harnesses measuring the capabilities, domains, and failure modes that matter to your model, plus independent scoring on public benchmarks.",
      "pricingNote": "",
      "availability": "available",
      "url": "https://www.appen.com/llm-training-data/llm-evaluation-benchmarks"
    },
    {
      "name": "Enterprise context engineering",
      "description": "Transform fragmented enterprise knowledge, workflows, and systems into secure, governed, agent-ready assets, scrubbed of sensitive data and structured for retrieval and action.",
      "pricingNote": "",
      "availability": "available",
      "url": "https://www.appen.com/agentic-rag-evaluation"
    },
    {
      "name": "Expert-in-the-loop evaluation",
      "description": "End-to-end measurement of how accurately agents retrieve, synthesize, and apply private enterprise knowledge at production scale.",
      "pricingNote": "",
      "availability": "available",
      "url": "https://www.appen.com/rlhf-ai"
    },
    {
      "name": "LLM-as-a-judge deployment",
      "description": "Deploy calibrated, auditable LLM judges for scoring, preference evaluation, and policy checks, validated against human raters with measured agreement.",
      "pricingNote": "",
      "availability": "available",
      "url": "https://www.appen.com/multilingual-llm-as-a-judge-managed-service"
    },
    {
      "name": "Model red teaming and safety evaluation",
      "description": "Adversarial testing and safety evaluation across modalities, run by vetted specialists to surface failure modes before release.",
      "pricingNote": "",
      "availability": "available",
      "url": "https://www.appen.com/red-teaming"
    },
    {
      "name": "Synthetic data generation",
      "description": "Controlled synthetic-data pipelines that expand coverage and target edge cases while preserving traceability, diversity, and documented quality standards.",
      "pricingNote": "",
      "availability": "available",
      "url": "https://www.appen.com/model-evaluation-and-integrity"
    }
  ]
}