{
  "schema": "interviewcoach.spark_job_match_leaderboard.v1",
  "generated_at_utc": "2026-05-28T13:20:51Z",
  "source_runs": [
    {
      "run_label": "spark-adaptive-situation-matrix-elastic-2026-05-28",
      "path": "ops/job-evals/spark-adaptive-situation-matrix-elastic-2026-05-28.json",
      "reports_loaded": 6,
      "failed_scenarios": null,
      "answers_submitted": null,
      "average_score": null
    },
    {
      "run_label": "spark-adaptive-situation-matrix-multijob-2026-05-28",
      "path": "ops/job-evals/spark-adaptive-situation-matrix-multijob-2026-05-28.json",
      "reports_loaded": 10,
      "failed_scenarios": null,
      "answers_submitted": null,
      "average_score": null
    },
    {
      "run_label": "spark-adaptive-situation-matrix-multijob-rerun-timeout600-2026-05-28",
      "path": "ops/job-evals/spark-adaptive-situation-matrix-multijob-rerun-timeout600-2026-05-28.json",
      "reports_loaded": 2,
      "failed_scenarios": null,
      "answers_submitted": null,
      "average_score": null
    },
    {
      "run_label": "spark-adaptive-company-batch2-2026-05-28",
      "path": "ops/job-evals/spark-adaptive-company-batch2-2026-05-28.json",
      "reports_loaded": 4,
      "failed_scenarios": null,
      "answers_submitted": null,
      "average_score": null
    },
    {
      "run_label": "spark-adaptive-company-batch2-rerun-timeout600-2026-05-28",
      "path": "ops/job-evals/spark-adaptive-company-batch2-rerun-timeout600-2026-05-28.json",
      "reports_loaded": 2,
      "failed_scenarios": null,
      "answers_submitted": null,
      "average_score": null
    },
    {
      "run_label": "spark-adaptive-range-lower-2026-05-28",
      "path": "ops/job-evals/spark-adaptive-range-lower-2026-05-28.json",
      "reports_loaded": 5,
      "failed_scenarios": null,
      "answers_submitted": null,
      "average_score": null
    },
    {
      "run_label": "spark-adaptive-range-lower-rerun-newrelic-timeout600-2026-05-28",
      "path": "ops/job-evals/spark-adaptive-range-lower-rerun-newrelic-timeout600-2026-05-28.json",
      "reports_loaded": 1,
      "failed_scenarios": null,
      "answers_submitted": null,
      "average_score": null
    },
    {
      "run_label": "spark-adaptive-range-hard-2026-05-28",
      "path": "ops/job-evals/spark-adaptive-range-hard-2026-05-28.json",
      "reports_loaded": 4,
      "failed_scenarios": null,
      "answers_submitted": null,
      "average_score": null
    }
  ],
  "summary": {
    "completed_reports": 34,
    "failed_or_incomplete_reports": 7,
    "answers_submitted": 170,
    "companies": 11,
    "jobs": 12,
    "average_score": 4.34,
    "min_score": 3.98,
    "max_score": 4.65
  },
  "company_ranking": [
    {
      "company": "Trase Systems",
      "reports": 2,
      "answers": 10,
      "average_score": 4.52,
      "min_score": 4.5,
      "max_score": 4.53
    },
    {
      "company": "Extreme Networks",
      "reports": 2,
      "answers": 10,
      "average_score": 4.42,
      "min_score": 4.38,
      "max_score": 4.45
    },
    {
      "company": "Bold Business",
      "reports": 2,
      "answers": 10,
      "average_score": 4.39,
      "min_score": 4.38,
      "max_score": 4.4
    },
    {
      "company": "Pinterest",
      "reports": 6,
      "answers": 30,
      "average_score": 4.37,
      "min_score": 4.25,
      "max_score": 4.6
    },
    {
      "company": "Elastic",
      "reports": 6,
      "answers": 30,
      "average_score": 4.36,
      "min_score": 4.2,
      "max_score": 4.5
    },
    {
      "company": "Grafana Labs",
      "reports": 4,
      "answers": 20,
      "average_score": 4.36,
      "min_score": 4.15,
      "max_score": 4.65
    },
    {
      "company": "OLX",
      "reports": 4,
      "answers": 20,
      "average_score": 4.36,
      "min_score": 4.2,
      "max_score": 4.47
    },
    {
      "company": "New Relic",
      "reports": 2,
      "answers": 10,
      "average_score": 4.31,
      "min_score": 4.2,
      "max_score": 4.42
    },
    {
      "company": "Finom",
      "reports": 2,
      "answers": 10,
      "average_score": 4.25,
      "min_score": 4.08,
      "max_score": 4.42
    },
    {
      "company": "Warp",
      "reports": 2,
      "answers": 10,
      "average_score": 4.18,
      "min_score": 4.02,
      "max_score": 4.35
    },
    {
      "company": "Sezzle",
      "reports": 2,
      "answers": 10,
      "average_score": 4.16,
      "min_score": 3.98,
      "max_score": 4.33
    }
  ],
  "job_ranking": [
    {
      "rank": 24,
      "company": "Trase Systems",
      "title": "Principal Applied ML Researcher (Agentic Systems & Applied AI Platform)",
      "job_url": "https://agentic-engineering-jobs.com/jobs/trase-systems-principal-applied-ml-researcher-agentic-systems-and-applied-ai-platform-OvCaPc",
      "apply_urls": [
        "https://job-boards.greenhouse.io/redcellpartners/jobs/5093368007"
      ],
      "reports": 2,
      "answers": 10,
      "average_score": 4.52,
      "min_score": 4.5,
      "max_score": 4.53
    },
    {
      "rank": 8,
      "company": "Pinterest",
      "title": "Principal Engineer, Agentic Engineering",
      "job_url": "https://agentic-engineering-jobs.com/jobs/pinterest-principal-engineer-agentic-engineering-BYrR1P",
      "apply_urls": [
        "https://www.pinterestcareers.com/jobs/?gh_jid=7775690"
      ],
      "reports": 4,
      "answers": 20,
      "average_score": 4.42,
      "min_score": 4.27,
      "max_score": 4.6
    },
    {
      "rank": 32,
      "company": "Extreme Networks",
      "title": "Principal Machine Learning Engineer-Gen AI, Machine Learning, Graph ML (10189)",
      "job_url": "https://agentic-engineering-jobs.com/jobs/extreme-networks-principal-machine-learning-engineer-gen-ai-machine-learning-graph-ml-10189-IQ7Ahs",
      "apply_urls": [
        "https://jobs.lever.co/extremenetworks/f99032c4-6048-484d-b93e-f8dfbb62aa19"
      ],
      "reports": 2,
      "answers": 10,
      "average_score": 4.42,
      "min_score": 4.38,
      "max_score": 4.45
    },
    {
      "rank": 9,
      "company": "Bold Business",
      "title": "Lead AI Engineer & Technical Architect (Remote)",
      "job_url": "https://agentic-engineering-jobs.com/jobs/bold-business-lead-ai-engineer-and-technical-architect-remote-np_4-_",
      "apply_urls": [
        "https://job-boards.greenhouse.io/boldbusiness/jobs/4239340009"
      ],
      "reports": 2,
      "answers": 10,
      "average_score": 4.39,
      "min_score": 4.38,
      "max_score": 4.4
    },
    {
      "rank": 1,
      "company": "Elastic",
      "title": "Elastic AI Engineer",
      "job_url": "https://agentic-engineering-jobs.com/jobs/elastic-elastic-ai-engineer-wBN5pn",
      "apply_urls": [
        "https://jobs.elastic.co/jobs?gh_jid=7858138"
      ],
      "reports": 6,
      "answers": 30,
      "average_score": 4.36,
      "min_score": 4.2,
      "max_score": 4.5
    },
    {
      "rank": 2,
      "company": "Grafana Labs",
      "title": "Staff AI Engineer | US | Remote (Marketing Ops)",
      "job_url": "https://agentic-engineering-jobs.com/jobs/grafana-labs-staff-ai-engineer-or-us-or-remote-marketing-ops-Hg9k3P",
      "apply_urls": [
        "https://job-boards.greenhouse.io/grafanalabs/jobs/5806328004"
      ],
      "reports": 4,
      "answers": 20,
      "average_score": 4.36,
      "min_score": 4.15,
      "max_score": 4.65
    },
    {
      "rank": 6,
      "company": "OLX",
      "title": "Lead Agentic AI Engineer",
      "job_url": "https://agentic-engineering-jobs.com/jobs/olx-lead-agentic-ai-engineer-q0aC8q",
      "apply_urls": [
        "https://jobs.eu.lever.co/olx/118bd6a4-a5af-4f99-98bb-7f51c08736d1"
      ],
      "reports": 4,
      "answers": 20,
      "average_score": 4.36,
      "min_score": 4.2,
      "max_score": 4.47
    },
    {
      "rank": 20,
      "company": "New Relic",
      "title": "AI Engineer",
      "job_url": "https://agentic-engineering-jobs.com/jobs/new-relic-ai-engineer-0aqhGD",
      "apply_urls": [
        "https://job-boards.greenhouse.io/newrelic/jobs/4987225008"
      ],
      "reports": 2,
      "answers": 10,
      "average_score": 4.31,
      "min_score": 4.2,
      "max_score": 4.42
    },
    {
      "rank": 33,
      "company": "Pinterest",
      "title": "AI Solutions Engineer",
      "job_url": "https://agentic-engineering-jobs.com/jobs/pinterest-ai-solutions-engineer-qKAOO_",
      "apply_urls": [
        "https://www.pinterestcareers.com/jobs/?gh_jid=7714127"
      ],
      "reports": 2,
      "answers": 10,
      "average_score": 4.26,
      "min_score": 4.25,
      "max_score": 4.27
    },
    {
      "rank": 7,
      "company": "Finom",
      "title": "Senior AI Engineer",
      "job_url": "https://agentic-engineering-jobs.com/jobs/finom-senior-ai-engineer-FfBKkQ",
      "apply_urls": [
        "https://jobs.eu.lever.co/pnlfin/733b12b7-c794-42d0-89f5-fcc24061ef0a"
      ],
      "reports": 2,
      "answers": 10,
      "average_score": 4.25,
      "min_score": 4.08,
      "max_score": 4.42
    },
    {
      "rank": 11,
      "company": "Warp",
      "title": "Forward Deployed Engineer",
      "job_url": "https://agentic-engineering-jobs.com/jobs/warp-forward-deployed-engineer-Ya4g2f",
      "apply_urls": [
        "https://job-boards.greenhouse.io/warp/jobs/5749183004"
      ],
      "reports": 2,
      "answers": 10,
      "average_score": 4.18,
      "min_score": 4.02,
      "max_score": 4.35
    },
    {
      "rank": 26,
      "company": "Sezzle",
      "title": "AI Engineering I - Marketing",
      "job_url": "https://agentic-engineering-jobs.com/jobs/sezzle-ai-engineering-i-marketing-H2y1Gp",
      "apply_urls": [
        "https://job-boards.greenhouse.io/sezzle/jobs/7709978003"
      ],
      "reports": 2,
      "answers": 10,
      "average_score": 4.16,
      "min_score": 3.98,
      "max_score": 4.33
    }
  ],
  "situation_ranking": [
    {
      "situation_id": "distributed-cache-hld-lld-tdd",
      "situation_title": "Distributed cache HLD/LLD plus TDD",
      "reports": 1,
      "answers": 5,
      "average_score": 4.5,
      "min_score": 4.5,
      "max_score": 4.5
    },
    {
      "situation_id": "streaming-storage-design-evolution",
      "situation_title": "Streaming/storage design evolution",
      "reports": 3,
      "answers": 15,
      "average_score": 4.46,
      "min_score": 4.27,
      "max_score": 4.65
    },
    {
      "situation_id": "ai-concepts-quickfire",
      "situation_title": "AI concepts quickfire round",
      "reports": 1,
      "answers": 5,
      "average_score": 4.45,
      "min_score": 4.45,
      "max_score": 4.45
    },
    {
      "situation_id": "craft-demo-live-repo-api",
      "situation_title": "Live repository/API craft demonstration",
      "reports": 1,
      "answers": 5,
      "average_score": 4.4,
      "min_score": 4.4,
      "max_score": 4.4
    },
    {
      "situation_id": "multi-agent-tool-safety",
      "situation_title": "Agentic workflow tool-safety review",
      "reports": 5,
      "answers": 25,
      "average_score": 4.38,
      "min_score": 4.15,
      "max_score": 4.5
    },
    {
      "situation_id": "langgraph-support-agent-build",
      "situation_title": "LangGraph support agent take-home defense",
      "reports": 3,
      "answers": 15,
      "average_score": 4.38,
      "min_score": 4.35,
      "max_score": 4.42
    },
    {
      "situation_id": "project-deep-dive-business-context",
      "situation_title": "Project deep dive with business context",
      "reports": 8,
      "answers": 40,
      "average_score": 4.37,
      "min_score": 4.2,
      "max_score": 4.6
    },
    {
      "situation_id": "coding-hint-recovery",
      "situation_title": "Coding round recovery after a hint",
      "reports": 1,
      "answers": 5,
      "average_score": 4.33,
      "min_score": 4.33,
      "max_score": 4.33
    },
    {
      "situation_id": "recruiter-screen-motivation-constraints",
      "situation_title": "Recruiter screen motivation and constraints",
      "reports": 6,
      "answers": 30,
      "average_score": 4.27,
      "min_score": 3.98,
      "max_score": 4.42
    },
    {
      "situation_id": "chaotic-process-recovery",
      "situation_title": "Chaotic interview process recovery",
      "reports": 1,
      "answers": 5,
      "average_score": 4.25,
      "min_score": 4.25,
      "max_score": 4.25
    },
    {
      "situation_id": "leadership-principles-unique-stories",
      "situation_title": "Leadership-principles story bank pressure",
      "reports": 1,
      "answers": 5,
      "average_score": 4.2,
      "min_score": 4.2,
      "max_score": 4.2
    },
    {
      "situation_id": "rag-operational-failure",
      "situation_title": "RAG operational failure drill",
      "reports": 3,
      "answers": 15,
      "average_score": 4.17,
      "min_score": 4.02,
      "max_score": 4.4
    }
  ],
  "top_reports": [
    {
      "run_label": "spark-adaptive-situation-matrix-multijob-2026-05-28",
      "rank": 2,
      "company": "Grafana Labs",
      "title": "Staff AI Engineer | US | Remote (Marketing Ops)",
      "job_url": "https://agentic-engineering-jobs.com/jobs/grafana-labs-staff-ai-engineer-or-us-or-remote-marketing-ops-Hg9k3P",
      "apply_urls": [
        "https://job-boards.greenhouse.io/grafanalabs/jobs/5806328004"
      ],
      "situation_id": "streaming-storage-design-evolution",
      "situation_title": "Streaming/storage design evolution",
      "turns_submitted": 5,
      "average_score": 4.65,
      "report_loaded": true,
      "report_summary": "Iurii presents as a strong, production-oriented hire candidate for a Staff AI Engineer Marketing Ops role. The answer set shows deep technical ownership and safe rollout discipline, with clear reliability and quality controls. Main hiring confidence blockers are mainly business-impact storytelling and operating/ownership detail, not core competence.",
      "strengths": [
        "Demonstrates ownership of a complete operating system end-to-end: immutable `lead_id` source-of-truth, explicit state transitions, lease/version checks, and idempotent retries to prevent duplicate writes.",
        "Shows clear production-grade delivery rigor through staged rollout, feature-flaged control, rollback conditions, and manual-review escape hatches, indicating reliability-first decision-making under risk.",
        "Frames work in business-relevant terms for RevOps/SDR: first-pass handoff correctness, reassignment rate, p95 latency-to-assign, and SLA adherence rather than only model metrics."
      ],
      "gaps": [
        "Business impact is mostly operational and technical; lacks explicit outcome baseline-to-target examples (e.g., projected revenue/ARR uplift, pipeline quality improvement, SDR throughput gain) that make hireability for marketing-ops leadership more obvious.",
        "Technology stack choices are broad but somewhat unconsolidated (`Node.js`, `Node-RED`, `n8n`, Python) without a clear rationale for maintainability and ownership boundaries at team scale.",
        "Cross-functional ownership is defined for RevOps/SDR but still light on communication mechanics (cadence, escalation path, KPI review rhythm), so repeated recruiter concern is leadership communication execution, not technical intent."
      ],
      "next_steps": [
        "Convert each strongest technical concept into a concise recruiter-ready story: Problem → approach → metric impact → ownership handoff (e.g., \"reduced duplicate routing from X% to Y% in first 2 weeks\").",
        "Add explicit ownership model in people terms: who owns policy logic, retrieval quality, infra reliability, alert triage, and customer-facing rollout approvals; mention your operating cadence with RevOps/SDR leadership.",
        "Anchor to business outcomes with concrete goals: qualified lead quality, SDR capacity, conversion-to-opportunity lift, and cost-to-handle per lead, with a clear before/after target range."
      ],
      "failure_cause": null
    },
    {
      "run_label": "spark-adaptive-situation-matrix-multijob-2026-05-28",
      "rank": 8,
      "company": "Pinterest",
      "title": "Principal Engineer, Agentic Engineering",
      "job_url": "https://agentic-engineering-jobs.com/jobs/pinterest-principal-engineer-agentic-engineering-BYrR1P",
      "apply_urls": [
        "https://www.pinterestcareers.com/jobs/?gh_jid=7775690"
      ],
      "situation_id": "project-deep-dive-business-context",
      "situation_title": "Project deep dive with business context",
      "turns_submitted": 5,
      "average_score": 4.6,
      "report_loaded": true,
      "report_summary": "The candidate presents as a credible principal-level engineering hire for an agentic platform role with strong safety-first architecture instincts and measurable operating discipline. They are explicit about ownership, guardrails, and decision gates, and use concrete thresholds to justify rollout pace. Main reservation is proof and business context: they explain what to do very well, but do not consistently tie it to hard past outcomes, broader org influence, or principal-level leadership mechanics beyond platform governance.",
      "strengths": [
        "Very clear ownership framing: they propose one concrete 90-day production outcome and name decision owners (single platform release owner, platform/security/CI-CD council), which increases confidence in accountability.",
        "Strong control-plane/system design posture for Pinterest scale: tenant isolation, explicit contracts (input/state/memory/escalation), replayable graphs, and policy wrappers for model calls indicate production maturity.",
        "Excellent measurable safety discipline: uses explicit gates and rollback criteria (e.g., unsafe-action escape >1.0% abs or >2x rolling 7-day baseline; 2.0% or 3x for two windows triggers rollback review), showing defensible decision criteria."
      ],
      "gaps": [
        "No quantified business impact from past execution is given beyond strategic statements; no hard proof like incident reduction, adoption lift, or cycle-time gains from prior work, so credibility is claimed more than evidenced.",
        "Timeline and scope are somewhat inconsistent: they describe both 30-day and 90-day plans with broad references to 3,000+ engineers, but stop short of a realistic phased delivery plan and staffing/capacity assumptions.",
        "Principal-leadership signals (cross-team influence, alignment with product goals, mentoring, and priority tradeoff discussions with non-technical stakeholders) are implied but not demonstrated as past behavior."
      ],
      "next_steps": [
        "Quantify ownership with recruiter-ready proof: provide a 2–3 sentence story from Sell.Systems with before/after metrics (e.g., incidents, cycle-time, adoption, rework/revert reduction) directly tied to business impact.",
        "Turn the 30/90-day plans into a strict execution map: day 1–7 baseline, day 8–30 control-plane hardening, day 31–60 pilot cohort + criteria, day 61–90 expansion gates, with named collaborators and clear deliverables per week.",
        "Address missing principal-level layer explicitly: describe how you align with product leadership, define roadmap tradeoffs, and drive org change across 3,000+ engineers without becoming a single-project hero story."
      ],
      "failure_cause": null
    },
    {
      "run_label": "spark-adaptive-range-hard-2026-05-28",
      "rank": 24,
      "company": "Trase Systems",
      "title": "Principal Applied ML Researcher (Agentic Systems & Applied AI Platform)",
      "job_url": "https://agentic-engineering-jobs.com/jobs/trase-systems-principal-applied-ml-researcher-agentic-systems-and-applied-ai-platform-OvCaPc",
      "apply_urls": [
        "https://job-boards.greenhouse.io/redcellpartners/jobs/5093368007"
      ],
      "situation_id": "project-deep-dive-business-context",
      "situation_title": "Project deep dive with business context",
      "turns_submitted": 5,
      "average_score": 4.53,
      "report_loaded": true,
      "report_summary": "Strong systems design response with explicit decomposition, failure isolation, and control loops, aligned to production reliability concerns. Primary weakness is that the answer is operationally mature but still under-specifies global scale limits and ownership/HA details for critical control-plane components.",
      "strengths": [
        "High-quality service decomposition: explicitly separated orchestrator, vector/memory, tool execution, evaluator, and outbox/retry boundaries with dedicated queues, timeouts, retries, and policy semantics. This is a solid architectural boundary strategy for blast-radius control.",
        "Strong failure-containment mechanics: uses connector-level circuit breaking, idempotent adapters, DLQ handling, quarantine logic, and domain-specific rollback (canary-level rollback only), which shows clear thinking about fault isolation under production stress.",
        "Concrete correctness and state-safety controls: immutable intent IDs, outbox-before-exec, idempotency keys, and causal-write gates prevent duplicate side effects and aborted-tool contamination into memory/embeddings. This is directly aligned with production integrity."
      ],
      "gaps": [
        "Scale assumptions are not made explicit enough. The candidate gives thresholds (e.g., 15% failure, two five-minute windows) but does not define workload envelope (QPS, concurrency, session mix) or capacity planning inputs that justify those thresholds, so review risk judgments remain heuristic.",
        "Orchestrator is the control brain but there is no clear failover/availability design for it (active/active, leader election, partition tolerance, or degraded mode behavior if evaluator/control plane itself degrades). This leaves an implicit single point of operational failure.",
        "No explicit ownership model for domain-level incident response. The design lacks who owns what SLO, who can approve rollback/release gates, and how cross-team handoff is handled during domain quarantine."
      ],
      "next_steps": [
        "Produce an explicit architecture map (service graph + failure domains) showing synchronous/asynchronous edges, dependency direction, and failure propagation rules for orchestrator, outbox/retry, connectors, evaluator, and memory write/read lanes.",
        "Add a measurable scale model: per-domain load envelopes, queue capacity math, headroom formulas, and derived alert thresholds (not only fixed percentages) for p95/p99 latency, queue age, DLQ growth, and duplicate-action budget.",
        "Define control-plane HA and recovery playbooks: orchestrator/evaluator redundancy strategy, state replication, split-brain prevention, and fail-closed/degraded behaviors with owner-annotated runbooks for each domain."
      ],
      "failure_cause": null
    },
    {
      "run_label": "spark-adaptive-situation-matrix-elastic-2026-05-28",
      "rank": 1,
      "company": "Elastic",
      "title": "Elastic AI Engineer",
      "job_url": "https://agentic-engineering-jobs.com/jobs/elastic-elastic-ai-engineer-wBN5pn",
      "apply_urls": [
        "https://jobs.elastic.co/jobs?gh_jid=7858138"
      ],
      "situation_id": "distributed-cache-hld-lld-tdd",
      "situation_title": "Distributed cache HLD/LLD plus TDD",
      "turns_submitted": 5,
      "average_score": 4.5,
      "report_loaded": true,
      "report_summary": "The candidate presents a clearly production-oriented architecture for an AI workflow under load, with explicit service boundaries, queue-backed control flow, failure isolation, and rollback behavior. The design is strong on control loops and observability. Weaknesses are mostly in depth gaps around operating assumptions, cross-cutting durability/DR, and hardening of control-plane/runbook mechanics beyond the steady-state design.",
      "strengths": [
        "Architectural boundaries are explicit and consistent across all turns: Ingestion/API, Graph Orchestrator, RAG Memory, Action Router, and per-system adapters, with immutable/versioned contracts between boundaries.",
        "Demonstrates scale-aware control: durable queueing, tenant-aware admission control, per-tenant token buckets, and queue-depth/backpressure tied to model throughput budgets at ~10k+ req/min.",
        "Implements concrete failure isolation patterns: bulkheads per adapter, DLQs at multiple async boundaries, bounded retries, adapter circuit breakers, and per-boundary rollback rather than global fail-open/-down."
      ],
      "gaps": [
        "No explicit data-plane durability and recovery model beyond 'durable queue' (e.g., broker HA settings, retention, replay limits, dedupe store TTL/eviction, and backup/restore strategy).",
        "Cross-region/zone resilience is not addressed; recovery and failover behavior is mostly local boundary-oriented and does not prove regional fault tolerance under datacenter-level outages.",
        "Model/provider failure handling is underspecified: no clear fallback across model versions/providers, max cost/latency tradeoff policy, or budgeted degrade behavior when LLM inference itself becomes the bottleneck."
      ],
      "next_steps": [
        "Produce a boundary map with data/control planes: for each service define ownership, primary datastore, outbound dependencies, rate limits, and SLO/error budget. Include a failure-domain table (node, symptom, blast radius, auto-action, manual escalation).",
        "Add explicit capacity assumptions and sizing math: queue throughput to concurrency mapping, p95 targets by stage, expected retries, and admission throttling thresholds derived from host and model token budgets.",
        "Define a concrete rollback/recovery runbook: canary entry/exit criteria, rollback precedence, replay safety for partially processed commands, and DR playbook for broker/orchestrator regional outage."
      ],
      "failure_cause": null
    },
    {
      "run_label": "spark-adaptive-range-hard-2026-05-28",
      "rank": 24,
      "company": "Trase Systems",
      "title": "Principal Applied ML Researcher (Agentic Systems & Applied AI Platform)",
      "job_url": "https://agentic-engineering-jobs.com/jobs/trase-systems-principal-applied-ml-researcher-agentic-systems-and-applied-ai-platform-OvCaPc",
      "apply_urls": [
        "https://job-boards.greenhouse.io/redcellpartners/jobs/5093368007"
      ],
      "situation_id": "multi-agent-tool-safety",
      "situation_title": "Agentic workflow tool-safety review",
      "turns_submitted": 5,
      "average_score": 4.5,
      "report_loaded": true,
      "report_summary": "Strong production-oriented systems design under safety and scale pressure. Candidate consistently reasons in terms of bounded services, strict failure boundaries, and automatic control loops, and ties partial-retrieval degradation to safe execution modes. Biggest weaknesses are the missing quantitative operational targets and explicit interface/ownership/runbook details that would make this review-ready for real platform implementation.",
      "strengths": [
        "Clear decomposition: proposes bounded services (ingress/policy/orchestration/retrieval/tool execution/HITL/audit) and later formalized tenant-local Evidence service + global policy/eval service, showing boundary discipline.",
        "Strong failure handling: defines partial-retrieval conservative mode with deterministic policy (no writes, no privileged calls, mandatory HITL, dry-run for effectful tools), plus rollback hooks and idempotency keys.",
        "Good blast-radius control: per-tenant bulkheads, queue partitions, backpressure, and tenant isolation during degradation to keep healthy tenants on throughput while limiting fault propagation."
      ],
      "gaps": [
        "No quantitative capacity/SLO plan: states p95/p99 and budgets as objectives but never provides concrete values for latency, throughput, queue depth, retry limits, or saturation/scale thresholds.",
        "Cross-service contracts are under-specified: while payload fields are listed, there is no explicit schema versioning, backward-compatibility policy, or fault/error semantics between services (e.g., transient vs. hard failures, retry ownership, poison-message handling).",
        "No explicit control-plane resilience plan: minimal discussion of dependency failures for global policy/eval service, regional isolation beyond tenant bulkheads, or state recovery/failover for event evidence during region outages."
      ],
      "next_steps": [
        "Produce a service map with API contracts: define `EvidenceCapsule`, `PolicyDecision`, and execution plan schemas including versioning, optionality, error codes, and idempotency semantics per endpoint.",
        "Create a concrete failure-domain matrix per service and region (retrieval/policy/orchestrator/execution/HITL), with explicit state transitions for NORMAL→DEGRADED→HITL→RECOVERING, and tenant-level blast-radius boundaries.",
        "Add numeric operational budgets: target RPS, p95/p99/p99.9 caps, queue limits, retry/backoff policy, circuit-breaker thresholds, and roll-forward/rollback triggers with concrete values and hysteresis."
      ],
      "failure_cause": null
    },
    {
      "run_label": "spark-adaptive-situation-matrix-multijob-rerun-timeout600-2026-05-28",
      "rank": 6,
      "company": "OLX",
      "title": "Lead Agentic AI Engineer",
      "job_url": "https://agentic-engineering-jobs.com/jobs/olx-lead-agentic-ai-engineer-q0aC8q",
      "apply_urls": [
        "https://jobs.eu.lever.co/olx/118bd6a4-a5af-4f99-98bb-7f51c08736d1"
      ],
      "situation_id": "streaming-storage-design-evolution",
      "situation_title": "Streaming/storage design evolution",
      "turns_submitted": 5,
      "average_score": 4.47,
      "report_loaded": true,
      "report_summary": "The candidate demonstrates strong architectural judgment for a high-traffic agentic platform under load and failure pressure. They consistently define boundaries, idempotent execution, and control loops tied to rollout safety, with concrete mechanisms (envelope contracts, outbox/inbox, DLQ, circuit breakers, and trace-driven rollback gates). Their design is production-minded in failure isolation and blast-radius control, but still lacks some explicit delivery and operations depth: capacity envelopes, ownership/operational handoff, schema migration strategy, and concrete non-functional targets for each domain.",
      "strengths": [
        "Clear domain decomposition and boundary control: separates ingress/orchestrator/tool/ storage layers and explicitly treats Support, Legal, and CRM as distinct failure domains with one-way tool effects.",
        "Strong explicit reliability design: idempotency keys, conditional writes, outbox/inbox, per-domain bounded queues, DLQs, timeout budgets, and circuit breakers form a defensible failure-containment strategy.",
        "Good retry/duplicate-safety behavior: operation_id-based guards and retriable/pending states prevent duplicate side effects during adapter failures, especially under spikes."
      ],
      "gaps": [
        "No explicit, quantified system sizing (expected throughput, concurrency caps per domain, storage retention, cost envelope) despite repeatedly referencing 5x spikes; this leaves capacity and cost-risk assumptions vague for production planning.",
        "Event contract mentions versioning but lacks a concrete compatibility plan (schema evolution, rolling consumer migrations, and dead-letter handling for incompatible schema versions).",
        "Limited operational ownership model: no explicit runbook roles, ownership mapping by domain, or escalation paths beyond generic incident-safe intent."
      ],
      "next_steps": [
        "Produce a concrete service map for OLX with explicit in/out contracts: ingress -> orchestrator -> domain adapters, including owner/team, read/write data stores, and failure domains for each hop.",
        "Define failure-domain-by-domain SLOs/SLIs before implementation (e.g., p95 orchestration latency, CRM action error budget, DLQ growth/hour, replay-debt age, retry amplification ratio) and codify automated gate thresholds for 10%→50%→100% rollout.",
        "Add schema and migration playbook: DomainEventV1 compatibility matrix, consumer rollout order, poison-message handling for schema mismatch, and rollback compatibility checks in preflight."
      ],
      "failure_cause": null
    },
    {
      "run_label": "spark-adaptive-situation-matrix-elastic-2026-05-28",
      "rank": 1,
      "company": "Elastic",
      "title": "Elastic AI Engineer",
      "job_url": "https://agentic-engineering-jobs.com/jobs/elastic-elastic-ai-engineer-wBN5pn",
      "apply_urls": [
        "https://jobs.elastic.co/jobs?gh_jid=7858138"
      ],
      "situation_id": "ai-concepts-quickfire",
      "situation_title": "AI concepts quickfire round",
      "turns_submitted": 5,
      "average_score": 4.45,
      "report_loaded": true,
      "report_summary": "The candidate gave a strong architecture-focused design with explicit component boundaries, circuit-breaker behavior, and SLO-driven control loops that would be acceptable in a senior systems review. The strongest evidence is their repeated, consistent treatment of blast-radius control under latency/retrieval and connector failure modes. Major remaining risk is missing hard scale assumptions and executable ownership/runbook details needed for production handoff under load and outages.",
      "strengths": [
        "Defined a clean service decomposition (ingestion, orchestrator, retrieval, action workers) with clear duty separation and one-way contracts, reducing unintended coupling and side effects.",
        "Demonstrated solid failure reasoning: shard/slice-level breakers, tenant/task partitioning, safe-mode transitions, DLQ handling, and manual handoff for high-risk actions.",
        "Consistently applied idempotency and replay safety at task/action boundaries, including immutable identifiers and dedupe checks to prevent retry amplification."
      ],
      "gaps": [
        "Did not provide a quantified capacity model (target TPS/QPS, worker counts, queue depths, shard-to-throughput mapping, storage/index fanout), so it is hard to judge whether the design actually scales under business load.",
        "Failure domains are conceptual but not fully mapped into a service map/runbook format: orchestrator state store, observability stack, message broker, and deployment topology (single vs multi-region, failover ordering) are not specified.",
        "No explicit ownership model is described (team/service owners, alert ownership, on-call ownership for breaker decisions), which is critical for production operational control loops and incident response."
      ],
      "next_steps": [
        "Produce a concrete service map with control-plane vs data-plane flow, including dependency edges and failure propagation paths for each of: retrieval, orchestrator, connectors, webhook ingress, and checkpoint/audit store.",
        "Add explicit scale assumptions and budgets (e.g., per-tenant and global QPS, p95/p99 latency budgets per stage, bounded queue capacities, worker concurrency targets, token/cost budgets for retrieval) and validate against worst-case burst scenarios.",
        "Define an explicit failure-domain matrix (tenant, connector, index shard, region, orchestrator, state store) with pre-defined mitigation actions and ownership per domain."
      ],
      "failure_cause": null
    },
    {
      "run_label": "spark-adaptive-range-hard-2026-05-28",
      "rank": 32,
      "company": "Extreme Networks",
      "title": "Principal Machine Learning Engineer-Gen AI, Machine Learning, Graph ML (10189)",
      "job_url": "https://agentic-engineering-jobs.com/jobs/extreme-networks-principal-machine-learning-engineer-gen-ai-machine-learning-graph-ml-10189-IQ7Ahs",
      "apply_urls": [
        "https://jobs.lever.co/extremenetworks/f99032c4-6048-484d-b93e-f8dfbb62aa19"
      ],
      "situation_id": "multi-agent-tool-safety",
      "situation_title": "Agentic workflow tool-safety review",
      "turns_submitted": 5,
      "average_score": 4.45,
      "report_loaded": true,
      "report_summary": "Strong systems-design performance under pressure. The candidate gives a coherent, safety-first architecture with explicit boundaries, control-plane authority, tenant-aware degradation, and observable failure control loops. For a principal ML platform role, the design quality is close to hireable, but some production details remain underspecified (scale economics, HA/DR, migration/runbook mechanics).",
      "strengths": [
        "The decomposition is explicit and operationally meaningful: orchestration/control plane, retrieval, inference, and telemetry are separate blast-radius domains with strict ownership and one-way dependency constraints, preventing cascading failure.",
        "Clear failure containment strategy appears repeatedly: vector-miss spikes trigger bounded fallback to graph/metadata, tenant-local circuiting exists, and unsafe tooling is explicitly gated by human approval/idempotency during degraded states.",
        "The candidate defines control-loop behaviors with concrete thresholds and state transitions (e.g., miss-ratio drift, p99 latency regression, timeout budget burn, and correlated multi-domain signal checks), then maps them to canary pause and rollback."
      ],
      "gaps": [
        "Tenant isolation and per-tenant circuiting are described, but there is no explicit service map of state ownership (e.g., where policy, tenant baselines, and circuit state are durably stored) and no HA/consistency model for the control plane.",
        "Thresholds are numerically plausible but not grounded by scale assumptions (RPS, request mix, model token budgets, memory/CPU headroom, queue length sizing, and expected burst shape), making capacity guarantees and SLO math incomplete.",
        "No explicit mention of deployment topology for regional resilience or failover (single-control-plane outage behavior, data-store quorum, control-loop persistence, or operator handoff sequence under control-plane partition)."
      ],
      "next_steps": [
        "Create an explicit architecture map with nodes, edges, ownership, and stateful components (control plane store, telemetry pipeline, queue brokers, model serving clusters), and mark failure boundaries and blast-radius scopes for each.",
        "Turn threshold guidance into a concrete policy document with formulas and hysteresis: e.g., baseline window length, statistical guard (mean/σ), cooldown windows, minimum sample size, and precedence rules when signals disagree.",
        "Add full capacity and economics sizing: per-service throughput model, model-serving throughput/latency envelopes, autoscaling triggers, queue depth limits, and expected cost impact at 10x and 100x growth."
      ],
      "failure_cause": null
    },
    {
      "run_label": "spark-adaptive-company-batch2-rerun-timeout600-2026-05-28",
      "rank": 7,
      "company": "Finom",
      "title": "Senior AI Engineer",
      "job_url": "https://agentic-engineering-jobs.com/jobs/finom-senior-ai-engineer-FfBKkQ",
      "apply_urls": [
        "https://jobs.eu.lever.co/pnlfin/733b12b7-c794-42d0-89f5-fcc24061ef0a"
      ],
      "situation_id": "langgraph-support-agent-build",
      "situation_title": "LangGraph support agent take-home defense",
      "turns_submitted": 5,
      "average_score": 4.42,
      "report_loaded": true,
      "report_summary": "Strongly architecture-oriented answer with clear boundary-based design and explicit failure ownership, but it stays mostly conceptual and does not fully quantify production scale assumptions, shared infrastructure failure domains, or formal SLO/error-budget mechanics. For a senior AI systems review, it is good-but-not-complete: the core judgment is solid, while operational rigor needs to become measurable and executable.",
      "strengths": [
        "Strong explicit decomposition into deterministic services: intake, parser/enrichment, retrieval, tool layer, model worker, and decision/escalation with explicit contracts between stages.",
        "Demonstrates production-grade failure control: per-boundary timeouts/retries/DLQ, contract gates, rollback triggers, canary/shadow scoring, feature-flagged rollout, and manual-review safe mode.",
        "Clear ownership model for incidents: uses first-boundary contract breach in time-window as arbiter, with blame-correction for collateral degradations and immutable audit records containing threshold versions."
      ],
      "gaps": [
        "No concrete service map with interfaces, synchronous/asynchronous edges, and data contracts beyond verbal description; hard to validate blast-radius and dependency order in a real review.",
        "No explicit capacity or throughput baselines (e.g., RPS, payload sizes, vector dimensions, model tokens/sec, storage growth), so SLOs cannot be stress-tested for mobile onboarding load patterns.",
        "Critical shared dependencies are underspecified: queue broker, metadata store, auth/secrets, model serving infrastructure, and cache layers are not assigned failure domains."
      ],
      "next_steps": [
        "Add an explicit service map: nodes, data schemas, API contracts, dependency directions, and failure propagation paths for intake→parser→retrieval→LLM→policy.",
        "Quantify contract SLAs per boundary (e.g., parser p95/p99, recall floor, LLM latency budget, queue limits, acceptable DLQ growth rate) and tie them to error-budget-based rollout/rollback rules.",
        "Define explicit runbooks for each failure domain (parser, retrieval, model worker, orchestration) with ownership handoff, blast-radius checks, and required audit artifacts."
      ],
      "failure_cause": null
    },
    {
      "run_label": "spark-adaptive-range-lower-2026-05-28",
      "rank": 20,
      "company": "New Relic",
      "title": "AI Engineer",
      "job_url": "https://agentic-engineering-jobs.com/jobs/new-relic-ai-engineer-0aqhGD",
      "apply_urls": [
        "https://job-boards.greenhouse.io/newrelic/jobs/4987225008"
      ],
      "situation_id": "recruiter-screen-motivation-constraints",
      "situation_title": "Recruiter screen motivation and constraints",
      "turns_submitted": 5,
      "average_score": 4.42,
      "report_loaded": true,
      "report_summary": "Strong systems-oriented design performance with clear stage boundaries, deterministic replay semantics, and explicit operational guardrails. The candidate reasons in control-loop terms and repeatedly ties correctness to isolation, observability, and controlled rollout. Main weakness is limited concrete operational specification for production sizing, failure-domain ownership, and external dependency runbook behavior (LLM/serving stack), which leaves some ambiguity at production handoff.",
      "strengths": [
        "Defines a strong architecture boundary model: immutable Android raw intake, deterministic `event_id`/idempotency, explicit versioned metadata contracts, and schema-gated staging before semantic writes.",
        "Implements blast-radius containment well: per-domain queues/services for ingest, normalization, entity-resolution, and anomaly/RAG with independent checkpoints, rollback policies, and freeze-on-failure behavior.",
        "Demonstrates replay safety and incident recovery discipline: DLQ + quarantine, manifest-driven reprocessing, partition-scoped rollback keys (platform/app/time bucket/hash), and staged canary/full resume controls."
      ],
      "gaps": [
        "Scale assumptions are under-specified for New Relic-grade production load: no event throughput profile, cardinality expectations, queue retention windows, p95/p99 latency targets, or autoscaling policy tied to observed lag and cost/throughput trade-offs.",
        "Failure modes are partly incomplete for cross-cutting dependencies, especially LLM/rerieval/Vector tooling outages or latency explosions; no explicit fallback behavior when Gemini/LangChain/LlamaIndex dependencies degrade.",
        "Rollout safety lacks explicit decision ownership and human-in-the-loop semantics: who approves canary gates, who can override autonomous rollback, and what constitutes a safe automatic vs manual rollback path."
      ],
      "next_steps": [
        "Produce a service map artifact (ASCII/diagram) listing each service, queue/topic, schema, SLO, owner, and failure domain with explicit inputs/outputs and retry contracts.",
        "Create a failure-domain table that enumerates likely faults by stage (normalization bug, retrieval timeout, model drift, checkpoint corruption, timezone data source regression) and deterministic remediations with ownership and expected blast radius.",
        "Add concrete scale envelopes: target events/sec, burst factor, max acceptable lag, queue depth thresholds, and autoscaling/reactive throttling rules per stage; include throughput/latency budgets and cost caps for tokenized enrichment."
      ],
      "failure_cause": null
    },
    {
      "run_label": "spark-adaptive-situation-matrix-elastic-2026-05-28",
      "rank": 1,
      "company": "Elastic",
      "title": "Elastic AI Engineer",
      "job_url": "https://agentic-engineering-jobs.com/jobs/elastic-elastic-ai-engineer-wBN5pn",
      "apply_urls": [
        "https://jobs.elastic.co/jobs?gh_jid=7858138"
      ],
      "situation_id": "craft-demo-live-repo-api",
      "situation_title": "Live repository/API craft demonstration",
      "turns_submitted": 5,
      "average_score": 4.4,
      "report_loaded": true,
      "report_summary": "As an architecture review lens, the candidate consistently reasoned in bounded service terms with explicit control ownership, immutable state, and operational rollback triggers. The design is strong on failure containment and idempotent side-effect control, but it is still missing a few production artifacts: explicit service maps, quantified scaling assumptions, and a formal runbook-style control loop for escalation and recovery.",
      "strengths": [
        "Clear boundary decomposition: Control (state owner), Policy (governance), Memory (retrieval), and Tool (side effects) with a single path to mutation, which is a robust blast-radius containment pattern.",
        "Strong failure-mode awareness: explicit saturation signaling, backpressure, circuit/token-bucket-style constraints, and short-circuiting under risk or low-confidence conditions rather than blind retries.",
        "Good replay and correctness protections: immutable event stream with correlation IDs/state_version and mutation fingerprinting, terminal-only ToolResult persistence, and idempotency to prevent duplicate writes under timeout ambiguity."
      ],
      "gaps": [
        "No explicit service-level contract map (API boundaries, request/response schemas, queue technologies, timeout contracts per hop) to turn the boundary design into an executable architecture diagram.",
        "Missing detailed failure domain matrix: no explicit handling for partial regional outages, ES replica failover behavior, or dependent service dependency graph under cascading failure.",
        "Retry policy is broad but under-specified operationally (only max_retry/thresholds shown); no jitter/backoff strategy per failure class, no differentiation between deterministic vs. non-deterministic operations."
      ],
      "next_steps": [
        "Produce an explicit service map/flow document with interface contracts: ingress → Control → Policy/Memory/Tool, including sync/async edges, payload versions, failure codes, and ownership at each hop.",
        "Define a failure-domain table (ES degradation, Tool timeout, Policy reject, orchestrator crash, duplicate replay) with precise transition actions, state transitions, and owner responsibility.",
        "Add hard scale model inputs: traffic shape, p50/p95/p99 latency budgets per stage, queue throughput, ES index/search QPS limits, autoscale thresholds, and saturation breakpoints tied to rollout policy."
      ],
      "failure_cause": null
    },
    {
      "run_label": "spark-adaptive-situation-matrix-multijob-2026-05-28",
      "rank": 8,
      "company": "Pinterest",
      "title": "Principal Engineer, Agentic Engineering",
      "job_url": "https://agentic-engineering-jobs.com/jobs/pinterest-principal-engineer-agentic-engineering-BYrR1P",
      "apply_urls": [
        "https://www.pinterestcareers.com/jobs/?gh_jid=7775690"
      ],
      "situation_id": "recruiter-screen-motivation-constraints",
      "situation_title": "Recruiter screen motivation and constraints",
      "turns_submitted": 5,
      "average_score": 4.4,
      "report_loaded": true,
      "report_summary": "Iurii is a strong technical communicator with a rigorous, safety-first design approach, but he is not yet fully convincing as a Principal Engineer hire at this stage because evidence is strategy-heavy and lacks concrete, past execution ownership and business-outcome proof at scale.",
      "strengths": [
        "Consistently articulates a production-minded architecture (control/execution/governance planes) aligned to safety, scalability, and auditability for a 3,000-engineer environment.",
        "Shows disciplined rollout thinking with staged autonomy, clear gates, and explicit trust-building controls (human-in-the-loop, policy routing, rollback requirements).",
        "Uses measurable decision signals rather than hype: first-pass accepted merge rate, agent-attributable post-merge escape rate, cost/latency telemetry, and confidence windows before promotion."
      ],
      "gaps": [
        "No concrete ownership evidence in the transcript: no specific team size, initiative leadership, delivery ownership, or direct examples of work shipped end-to-end (ownership signals are effectively absent).",
        "Business impact is aspirational, not evidenced: targets are proposed but not grounded in historical outcomes, baselines, or quantifiable business KPIs from actual programs.",
        "Cross-functional communication is defined structurally (safety trio, owners) but not demonstrated through real examples of conflict resolution, prioritization tradeoffs with product/security, or stakeholder alignment under pressure."
      ],
      "next_steps": [
        "Reframe the strongest answers into STAR-style, production stories: problem, concrete actions, measurable results, and why it mattered to customers/business; include what changed after his previous architecture rollout.",
        "Add two concrete ownership proofs immediately in interview: one architecture initiative he led, one platform policy/safety program he owned, each with quantified impact (e.g., defects, rollout time, cost, latency, incident reduction).",
        "State explicit collaboration outcomes: who he influenced, who resisted, how he got sign-off, and how he managed tradeoffs with security/product/engineering leadership."
      ],
      "failure_cause": null
    }
  ],
  "failures": [
    {
      "run_label": "spark-adaptive-situation-matrix-multijob-2026-05-28",
      "rank": 6,
      "company": "OLX",
      "title": "Lead Agentic AI Engineer",
      "job_url": "https://agentic-engineering-jobs.com/jobs/olx-lead-agentic-ai-engineer-q0aC8q",
      "apply_urls": [],
      "situation_id": "streaming-storage-design-evolution",
      "situation_title": "Streaming/storage design evolution",
      "turns_submitted": null,
      "average_score": null,
      "report_loaded": false,
      "report_summary": null,
      "strengths": [],
      "gaps": [],
      "next_steps": [],
      "failure_cause": "Spark usage/rate limit reached."
    },
    {
      "run_label": "spark-adaptive-situation-matrix-multijob-2026-05-28",
      "rank": 8,
      "company": "Pinterest",
      "title": "Principal Engineer, Agentic Engineering",
      "job_url": "https://agentic-engineering-jobs.com/jobs/pinterest-principal-engineer-agentic-engineering-BYrR1P",
      "apply_urls": [],
      "situation_id": "streaming-storage-design-evolution",
      "situation_title": "Streaming/storage design evolution",
      "turns_submitted": null,
      "average_score": null,
      "report_loaded": false,
      "report_summary": null,
      "strengths": [],
      "gaps": [],
      "next_steps": [],
      "failure_cause": "Spark dependency unavailable. request_id=5ce88c6b2fdc45ba8af5973a85d721ef"
    },
    {
      "run_label": "spark-adaptive-company-batch2-2026-05-28",
      "rank": 7,
      "company": "Finom",
      "title": "Senior AI Engineer",
      "job_url": "https://agentic-engineering-jobs.com/jobs/finom-senior-ai-engineer-FfBKkQ",
      "apply_urls": [],
      "situation_id": "langgraph-support-agent-build",
      "situation_title": "LangGraph support agent take-home defense",
      "turns_submitted": null,
      "average_score": null,
      "report_loaded": false,
      "report_summary": null,
      "strengths": [],
      "gaps": [],
      "next_steps": [],
      "failure_cause": "Spark dependency unavailable. request_id=e3d0beac819d47a7b24b95762f22d458"
    },
    {
      "run_label": "spark-adaptive-company-batch2-2026-05-28",
      "rank": 11,
      "company": "Warp",
      "title": "Forward Deployed Engineer",
      "job_url": "https://agentic-engineering-jobs.com/jobs/warp-forward-deployed-engineer-Ya4g2f",
      "apply_urls": [],
      "situation_id": "langgraph-support-agent-build",
      "situation_title": "LangGraph support agent take-home defense",
      "turns_submitted": null,
      "average_score": null,
      "report_loaded": false,
      "report_summary": null,
      "strengths": [],
      "gaps": [],
      "next_steps": [],
      "failure_cause": "Spark dependency unavailable. request_id=68f9279d001341ca81d06a6988e59356"
    },
    {
      "run_label": "spark-adaptive-range-lower-2026-05-28",
      "rank": 20,
      "company": "New Relic",
      "title": "AI Engineer",
      "job_url": "https://agentic-engineering-jobs.com/jobs/new-relic-ai-engineer-0aqhGD",
      "apply_urls": [],
      "situation_id": "project-deep-dive-business-context",
      "situation_title": "Project deep dive with business context",
      "turns_submitted": null,
      "average_score": null,
      "report_loaded": false,
      "report_summary": null,
      "strengths": [],
      "gaps": [],
      "next_steps": [],
      "failure_cause": "Spark stream disconnected before completion. request_id=db516430-8b56-4bc6-ab31-8ef8f561bdaf"
    },
    {
      "run_label": "spark-adaptive-range-hard-2026-05-28",
      "rank": 21,
      "company": "Natera",
      "title": "Director, Forward Deployed AI Engineering",
      "job_url": "https://agentic-engineering-jobs.com/jobs/natera-director-forward-deployed-ai-engineering-Btljx8",
      "apply_urls": [],
      "situation_id": "project-deep-dive-business-context",
      "situation_title": "Project deep dive with business context",
      "turns_submitted": null,
      "average_score": null,
      "report_loaded": false,
      "report_summary": null,
      "strengths": [],
      "gaps": [],
      "next_steps": [],
      "failure_cause": "Spark dependency unavailable. request_id=545938d91663415a860a63250e960d63"
    },
    {
      "run_label": "spark-adaptive-range-hard-2026-05-28",
      "rank": 21,
      "company": "Natera",
      "title": "Director, Forward Deployed AI Engineering",
      "job_url": "https://agentic-engineering-jobs.com/jobs/natera-director-forward-deployed-ai-engineering-Btljx8",
      "apply_urls": [],
      "situation_id": "multi-agent-tool-safety",
      "situation_title": "Agentic workflow tool-safety review",
      "turns_submitted": null,
      "average_score": null,
      "report_loaded": false,
      "report_summary": null,
      "strengths": [],
      "gaps": [],
      "next_steps": [],
      "failure_cause": "Spark dependency unavailable. request_id=60ebc15f9a6140359c8cc6610323504f"
    }
  ]
}
