Open frameworkJev QuadrantLive map

Agent observability and guardrails / entity

Galileo

Evaluation, observability, and runtime protection for GenAI applications.

https://www.galileo.ai

Proven Execution

0.368

Validated Direction

0.292

Uncertainty X

0.375

Uncertainty Y

0.380

Evidence mass

0.996

Region

OPERATIONAL

Quadrant distribution

Jev `choice` — not a hard 0.5 cut. Confidence 0.31.

  • Directed27.0%
  • Anchored4.0%
  • Forming20.0%
  • Operational49.0%

Evidence ledger

Recency half-life 180 days. Weight = 2^(-age/180).

  1. independent · accessed 2026-09-24 · w=0.996

    Hallucination detection tools 2026 (Galileo section)

    A Braintrust article describes Galileo’s hallucination workflow as Luna-2 evaluators, metrics such as Correctness and Context Adherence, and runtime guardrails with sub-200ms inline blocking on high-risk responses. This is a third-party description, not a Galileo primary source.

Reproducibility

resolved model: jev-1.13.0
requested model: jev-latest
request id: req_01a0d5e195c77a8883d152eb85234503
timestamp: 2026-09-25T00:05:17.859Z
request hash: 41227cefc0c7b4c9b3bdca1d29709d1d4ff8b091c0d7a3c8c3048408db28b3f8
Download request JSON
{
  "model": "jev-latest",
  "state": {
    "goal": "Score this entity on the Jev Quadrant rubric using only the supplied evidence.",
    "framework": "jev-quadrant-v1",
    "topic": {
      "id": "agent-observability",
      "title": "Agent observability and guardrails",
      "x_axis": "Proven Execution",
      "y_axis": "Validated Direction"
    },
    "entity": {
      "id": "galileo",
      "name": "Galileo",
      "website": "https://www.galileo.ai",
      "description": "Evaluation, observability, and runtime protection for GenAI applications."
    },
    "evidence": [
      {
        "id": "galileo-braintrust-roundup",
        "kind": "independent",
        "url": "https://www.braintrust.dev/articles/best-hallucination-detection-tools-2026",
        "date": "2026-09-24",
        "date_kind": "accessed",
        "recency_weight": 0.9962,
        "summary": "A Braintrust article describes Galileo’s hallucination workflow as Luna-2 evaluators, metrics such as Correctness and Context Adherence, and runtime guardrails with sub-200ms inline blocking on high-risk responses. This is a third-party description, not a Galileo primary source.",
        "title": "Hallucination detection tools 2026 (Galileo section)"
      }
    ],
    "claims": [],
    "independent_checks": [],
    "recency": {
      "half_life_days": 180,
      "as_of": "2026-09-25",
      "mass": 0.9962
    }
  },
  "questions": {
    "exec_shipped": {
      "type": "noul",
      "instructions": "Use only the supplied state. Do not use prior knowledge that is not represented as evidence items. If evidence is thin, return a probability near 0.5 with the understanding that uncertainty is handled separately. Has this entity shipped a generally available product used by third parties?",
      "criteria": {
        "true": "Independent public evidence shows a generally available product used by third parties.",
        "false": "The product is announced, preview-only, or evidenced only by unsourced vendor marketing with no third-party use."
      }
    },
    "exec_adoption": {
      "type": "noul",
      "instructions": "Use only the supplied state. Do not use prior knowledge that is not represented as evidence items. If evidence is thin, return a probability near 0.5 with the understanding that uncertainty is handled separately. Is there credible adoption evidence?",
      "criteria": {
        "true": "Credible adoption evidence exists: named customers, published user or download counts, case studies, or equivalent.",
        "false": "Adoption claims are absent, purely anecdotal, or unsourced."
      }
    },
    "exec_reliability": {
      "type": "noul",
      "instructions": "Use only the supplied state. Do not use prior knowledge that is not represented as evidence items. If evidence is thin, return a probability near 0.5 with the understanding that uncertainty is handled separately. Is there evidence of reliability, quality, security/compliance, or operational maturity?",
      "criteria": {
        "true": "Evidence of reliability, quality, security/compliance, or operational maturity (uptime, incidents handled, certifications, long-running open-source project).",
        "false": "No reliability or quality evidence is present."
      }
    },
    "exec_cadence": {
      "type": "noul",
      "instructions": "Use only the supplied state. Do not use prior knowledge that is not represented as evidence items. If evidence is thin, return a probability near 0.5 with the understanding that uncertainty is handled separately. Is there evidence of continued delivery after launch?",
      "criteria": {
        "true": "Evidence of continued delivery after launch (dated releases, changelog, subsequent features).",
        "false": "Delivery appears one-shot or stalled."
      }
    },
    "dir_strategy": {
      "type": "noul",
      "instructions": "Use only the supplied state. Do not use prior knowledge that is not represented as evidence items. If evidence is thin, return a probability near 0.5 with the understanding that uncertainty is handled separately. Is there a specific, consistent, dated public strategy or roadmap?",
      "criteria": {
        "true": "A public strategy or roadmap is specific, consistent, and dated.",
        "false": "Direction is vague, contradictory, or only implied."
      }
    },
    "dir_backed": {
      "type": "noul",
      "instructions": "Use only the supplied state. Do not use prior knowledge that is not represented as evidence items. If evidence is thin, return a probability near 0.5 with the understanding that uncertainty is handled separately. Are strategy or roadmap claims backed by shipped artifacts in the evidence pack?",
      "criteria": {
        "true": "Roadmap or strategy claims are backed by shipped artifacts in the evidence pack.",
        "false": "Direction is promised without corresponding delivery evidence."
      }
    },
    "dir_distinct": {
      "type": "noul",
      "instructions": "Use only the supplied state. Do not use prior knowledge that is not represented as evidence items. If evidence is thin, return a probability near 0.5 with the understanding that uncertainty is handled separately. Is the evidenced direction distinct from generic category copy?",
      "criteria": {
        "true": "The evidenced direction is distinct from generic category copy.",
        "false": "Positioning is interchangeable with peers."
      }
    },
    "dir_invested": {
      "type": "noul",
      "instructions": "Use only the supplied state. Do not use prior knowledge that is not represented as evidence items. If evidence is thin, return a probability near 0.5 with the understanding that uncertainty is handled separately. Is there evidence of sustained investment in this product?",
      "criteria": {
        "true": "Evidence of sustained investment (funding used for the product, research, open-source velocity, or platform expansion).",
        "false": "No evidence of continued investment."
      }
    },
    "quadrant": {
      "type": "choice",
      "instructions": "Use only the supplied state. Do not use prior knowledge that is not represented as evidence items. If evidence is thin, return a probability near 0.5 with the understanding that uncertainty is handled separately. Assign the entity to one Jev Quadrant cell. Do not use a hard 0.5 cut. Weigh the full evidence pack.",
      "criteria": {
        "ANCHORED": "Both delivery (proven execution) and evidenced direction (validated direction) are strong.",
        "DIRECTED": "Validated direction is better evidenced than proven execution.",
        "OPERATIONAL": "Proven execution is better evidenced than validated direction.",
        "FORMING": "Both axes are thin, early, or contradicted."
      }
    }
  }
}

Challenges

Anyone can submit a source. Payment never affects review.

Submit evidence for Galileo

No pending or accepted challenges yet.