Open frameworkJev QuadrantLive map

Agent observability and guardrails / entity

Arize Phoenix

Open-source AI observability with Phoenix Intelligence (PXI) agent.

https://arize.com/docs/phoenix

Proven Execution

0.378

Validated Direction

0.430

Uncertainty X

0.405

Uncertainty Y

0.356

Evidence mass

1.304

Region

OPERATIONAL

Quadrant distribution

Jev `choice` — not a hard 0.5 cut. Confidence 0.55.

  • Directed9.0%
  • Anchored1.0%
  • Operational67.0%
  • Forming23.0%

Evidence ledger

Recency half-life 180 days. Weight = 2^(-age/180).

  1. release · published 2026-06-02 · w=0.642

    Introducing PXI

    Arize Phoenix 17.0.0+ added PXI (Phoenix Intelligence), a beta AI engineering agent that investigates traces, prompts, and experiments in a sandboxed environment with filesystem, bash, and Phoenix MCP access.

  2. release · published 2026-06-10 · w=0.662

    PXI agent update

    Phoenix 17.3.0+ expanded PXI with a skills menu, parallel subagents, playground orchestration, evaluator authoring, dataset management, and Claude Fable 5 in the playground.

Reproducibility

resolved model: jev-1.13.0
requested model: jev-latest
request id: req_01a0d5e186bd7e2a8a52b1c09aba6834
timestamp: 2026-09-25T00:05:17.859Z
request hash: 13f28dd136f0435c0057cf710c9494ba7f8852e70d1669d8b3edf1ad562778b6
Download request JSON
{
  "model": "jev-latest",
  "state": {
    "goal": "Score this entity on the Jev Quadrant rubric using only the supplied evidence.",
    "framework": "jev-quadrant-v1",
    "topic": {
      "id": "agent-observability",
      "title": "Agent observability and guardrails",
      "x_axis": "Proven Execution",
      "y_axis": "Validated Direction"
    },
    "entity": {
      "id": "arize-phoenix",
      "name": "Arize Phoenix",
      "website": "https://arize.com/docs/phoenix",
      "description": "Open-source AI observability with Phoenix Intelligence (PXI) agent."
    },
    "evidence": [
      {
        "id": "pxi",
        "kind": "release",
        "url": "https://arize.com/docs/phoenix/release-notes/06-2026/06-02-2026-pxi-agent",
        "date": "2026-06-02",
        "date_kind": "published",
        "recency_weight": 0.6422,
        "summary": "Arize Phoenix 17.0.0+ added PXI (Phoenix Intelligence), a beta AI engineering agent that investigates traces, prompts, and experiments in a sandboxed environment with filesystem, bash, and Phoenix MCP access.",
        "title": "Introducing PXI"
      },
      {
        "id": "pxi-update",
        "kind": "release",
        "url": "https://arize.com/docs/phoenix/release-notes/06-2026/06-10-2026-pxi-agent-update",
        "date": "2026-06-10",
        "date_kind": "published",
        "recency_weight": 0.6623,
        "summary": "Phoenix 17.3.0+ expanded PXI with a skills menu, parallel subagents, playground orchestration, evaluator authoring, dataset management, and Claude Fable 5 in the playground.",
        "title": "PXI agent update"
      }
    ],
    "claims": [],
    "independent_checks": [],
    "recency": {
      "half_life_days": 180,
      "as_of": "2026-09-25",
      "mass": 1.3045
    }
  },
  "questions": {
    "exec_shipped": {
      "type": "noul",
      "instructions": "Use only the supplied state. Do not use prior knowledge that is not represented as evidence items. If evidence is thin, return a probability near 0.5 with the understanding that uncertainty is handled separately. Has this entity shipped a generally available product used by third parties?",
      "criteria": {
        "true": "Independent public evidence shows a generally available product used by third parties.",
        "false": "The product is announced, preview-only, or evidenced only by unsourced vendor marketing with no third-party use."
      }
    },
    "exec_adoption": {
      "type": "noul",
      "instructions": "Use only the supplied state. Do not use prior knowledge that is not represented as evidence items. If evidence is thin, return a probability near 0.5 with the understanding that uncertainty is handled separately. Is there credible adoption evidence?",
      "criteria": {
        "true": "Credible adoption evidence exists: named customers, published user or download counts, case studies, or equivalent.",
        "false": "Adoption claims are absent, purely anecdotal, or unsourced."
      }
    },
    "exec_reliability": {
      "type": "noul",
      "instructions": "Use only the supplied state. Do not use prior knowledge that is not represented as evidence items. If evidence is thin, return a probability near 0.5 with the understanding that uncertainty is handled separately. Is there evidence of reliability, quality, security/compliance, or operational maturity?",
      "criteria": {
        "true": "Evidence of reliability, quality, security/compliance, or operational maturity (uptime, incidents handled, certifications, long-running open-source project).",
        "false": "No reliability or quality evidence is present."
      }
    },
    "exec_cadence": {
      "type": "noul",
      "instructions": "Use only the supplied state. Do not use prior knowledge that is not represented as evidence items. If evidence is thin, return a probability near 0.5 with the understanding that uncertainty is handled separately. Is there evidence of continued delivery after launch?",
      "criteria": {
        "true": "Evidence of continued delivery after launch (dated releases, changelog, subsequent features).",
        "false": "Delivery appears one-shot or stalled."
      }
    },
    "dir_strategy": {
      "type": "noul",
      "instructions": "Use only the supplied state. Do not use prior knowledge that is not represented as evidence items. If evidence is thin, return a probability near 0.5 with the understanding that uncertainty is handled separately. Is there a specific, consistent, dated public strategy or roadmap?",
      "criteria": {
        "true": "A public strategy or roadmap is specific, consistent, and dated.",
        "false": "Direction is vague, contradictory, or only implied."
      }
    },
    "dir_backed": {
      "type": "noul",
      "instructions": "Use only the supplied state. Do not use prior knowledge that is not represented as evidence items. If evidence is thin, return a probability near 0.5 with the understanding that uncertainty is handled separately. Are strategy or roadmap claims backed by shipped artifacts in the evidence pack?",
      "criteria": {
        "true": "Roadmap or strategy claims are backed by shipped artifacts in the evidence pack.",
        "false": "Direction is promised without corresponding delivery evidence."
      }
    },
    "dir_distinct": {
      "type": "noul",
      "instructions": "Use only the supplied state. Do not use prior knowledge that is not represented as evidence items. If evidence is thin, return a probability near 0.5 with the understanding that uncertainty is handled separately. Is the evidenced direction distinct from generic category copy?",
      "criteria": {
        "true": "The evidenced direction is distinct from generic category copy.",
        "false": "Positioning is interchangeable with peers."
      }
    },
    "dir_invested": {
      "type": "noul",
      "instructions": "Use only the supplied state. Do not use prior knowledge that is not represented as evidence items. If evidence is thin, return a probability near 0.5 with the understanding that uncertainty is handled separately. Is there evidence of sustained investment in this product?",
      "criteria": {
        "true": "Evidence of sustained investment (funding used for the product, research, open-source velocity, or platform expansion).",
        "false": "No evidence of continued investment."
      }
    },
    "quadrant": {
      "type": "choice",
      "instructions": "Use only the supplied state. Do not use prior knowledge that is not represented as evidence items. If evidence is thin, return a probability near 0.5 with the understanding that uncertainty is handled separately. Assign the entity to one Jev Quadrant cell. Do not use a hard 0.5 cut. Weigh the full evidence pack.",
      "criteria": {
        "ANCHORED": "Both delivery (proven execution) and evidenced direction (validated direction) are strong.",
        "DIRECTED": "Validated direction is better evidenced than proven execution.",
        "OPERATIONAL": "Proven execution is better evidenced than validated direction.",
        "FORMING": "Both axes are thin, early, or contradicted."
      }
    }
  }
}

Challenges

Anyone can submit a source. Payment never affects review.

Submit evidence for Arize Phoenix

No pending or accepted challenges yet.