{
  "track_a": {
    "phase_summary": {
      "theoretical": 49,
      "empirical": 2,
      "applicable": 16,
      "replicable": 10,
      "impact": 60
    },
    "by_type": {
      "friction": 137,
      "absence": 0
    },
    "discoveries": {
      "A-S02-P00": {
        "phase": "impact",
        "type": "friction",
        "description": "Translation debt is critically high at 25.23, with a handoff failure rate of 64%. The sample shows a specific instance where AI output lost meaning at the approval handoff (decision 21), and multiple human settlement decisions (6, 10, 13, 16, 20) required translation=True, indicating meaning is being lost or re-interpreted at human-AI boundaries. The high exception rate (6.42%) and low first-pass accuracy (28%) suggest that translation failures are cascading into downstream errors.",
        "hypothesis": "The AI pipeline is producing outputs that are structurally incompatible with human workflow expectations. The 64% handoff failure rate suggests that AI outputs are being rejected or require re-interpretation at most handoff points, creating a systemic translation tax that degrades both accuracy and cycle time.",
        "confidence": 0.92,
        "improvement": 0.0,
        "consequence": 0.95,
        "evidence_sprints": [
          2
        ],
        "sprints_in_phase": 12
      },
      "A-S02-P01": {
        "phase": "impact",
        "type": "friction",
        "description": "There is a stark bimodal trust distribution. Jordan (ai_trust=1.00) and Kathryn (0.81) show high trust, while Diana (0.00), Tommy (0.00), and Nick (0.00) show zero trust. Diana has 26 negative AI experiences out of 28 total, Tommy has 15 negative out of 27, and both show high stress (0.30 and 0.63 respectively). The low-trust agents are making the majority of decisions (Diana: 72, Tommy: 75, Nick: 68), suggesting that the most experienced workers are actively rejecting AI assistance.",
        "hypothesis": "The high translation debt and handoff failures are directly poisoning trust. Agents like Diana and Tommy who interact with AI at complex decision points (settlement, subrogation) are experiencing repeated failures, while Jordan (who handles simple FNOL intake) sees consistent AI success. This creates a self-reinforcing cycle where high-expertise agents reject AI, forcing them to do more manual work, increasing exhaustion and stress.",
        "confidence": 0.88,
        "improvement": 0.0,
        "consequence": 0.9,
        "evidence_sprints": [
          2
        ],
        "sprints_in_phase": 12
      },
      "A-S02-P02": {
        "phase": "impact",
        "type": "friction",
        "description": "Diana has become a critical bottleneck with 72 decisions, exhaustion=8.0, and 26 negative AI experiences. She is handling settlement_human tasks that consistently require translation=True (decisions 6, 10, 13, 16, 20). Meanwhile, agents like Ron, Tricia, Mike, and Leslie have 0 decisions, suggesting work is not being distributed to available capacity. The cycle_time of 0.0 indicates the system may be stalled or metrics are not being captured properly.",
        "hypothesis": "Diana's expertise in settlement tasks, combined with her low AI trust, means she is manually processing complex claims that require translation. The system is not rebalancing work to idle agents (Ron, Tricia, Mike, Leslie) who have capacity. This creates a single point of failure where Diana's exhaustion directly impacts throughput.",
        "confidence": 0.85,
        "improvement": 0.0,
        "consequence": 0.85,
        "evidence_sprints": [
          2
        ],
        "sprints_in_phase": 12
      },
      "A-S02-P03": {
        "phase": "impact",
        "type": "friction",
        "description": "Exception rate is 6.42% with a specific pattern: decision 21 shows an AI approval exception where output lost meaning at handoff. The supplement_request_rate of 36% is extremely high, indicating that over a third of claims require additional information. This is coupled with a first_pass_accuracy of only 28%, meaning the AI pipeline is failing to process claims correctly on the first attempt.",
        "hypothesis": "The AI pipeline is generating incomplete or incorrect outputs that trigger exceptions and supplement requests. The 36% supplement rate suggests that AI is frequently requesting additional information that should have been captured earlier in the process, indicating a data completeness issue at intake or a failure in the AI's ability to synthesize available information.",
        "confidence": 0.82,
        "improvement": 0.0,
        "consequence": 0.8,
        "evidence_sprints": [
          2
        ],
        "sprints_in_phase": 12
      },
      "A-S02-P04": {
        "phase": "theoretical",
        "type": "friction",
        "description": "Kathryn (ai_trust=0.81, stress=0.05, exhaustion=5.0) is handling escalated exceptions (decision 22) despite having only 6 decisions total. She appears to be the designated escalation point, but her low decision count suggests she is underutilized while Diana is overloaded. Pat and Sanjay show high stress (1.00) with mixed AI experiences (+15/-11 and +23/-15 respectively), indicating they are in a 'trust struggle' phase.",
        "hypothesis": "The escalation path is informal and underutilized. Kathryn has high trust and low stress, making her an ideal escalation point, but the system is not routing enough work to her. Pat and Sanjay are experiencing decision fatigue from mixed AI experiences, which may be causing them to escalate more frequently or make inconsistent decisions.",
        "confidence": 0.7,
        "improvement": 0.0,
        "consequence": 0.65,
        "evidence_sprints": [
          2
        ],
        "sprints_in_phase": 15
      },
      "A-S02-P05": {
        "phase": "theoretical",
        "type": "friction",
        "description": "There is a clear pattern of AI adoption resistance among experienced agents. Diana (0.00), Tommy (0.00), and Nick (0.00) have zero trust, while Jordan (1.00) and Kathryn (0.81) show high trust. The zero-trust agents have high decision counts (72, 75, 68) and high exhaustion (8.0, 4.0, 9.0), suggesting they are working harder while rejecting AI assistance. Notably, Nick has +17 positive AI experiences and 0 negative, yet still shows 0.00 trust, indicating a fundamental resistance not based on experience.",
        "hypothesis": "Nick's case is particularly telling - he has only positive AI experiences but zero trust. This suggests a pre-existing bias against AI that is not experience-based. This could stem from job security concerns, a belief that human judgment is superior, or organizational culture that values manual expertise. The high exhaustion (9.0) suggests this resistance is costly to his wellbeing.",
        "confidence": 0.78,
        "improvement": 0.0,
        "consequence": 0.75,
        "evidence_sprints": [
          2
        ],
        "sprints_in_phase": 15
      },
      "A-S02-P06": {
        "phase": "theoretical",
        "type": "friction",
        "description": "The cycle_time of 0.0 combined with high decision counts (total 534 decisions across agents) and high exhaustion levels suggests the cycle_time metric is not capturing actual workflow duration. This is a metric inversion where the system reports zero cycle time while agents are clearly spending significant time on decisions (Diana has 72 decisions with exhaustion 8.0). The cost_per_claim of $424.48 with a 36% supplement rate suggests the true cycle time is substantial but not being measured.",
        "hypothesis": "The cycle_time metric is likely measuring only AI pipeline processing time, not the full human-in-the-loop cycle. Since many decisions require human intervention (especially with 64% handoff failure), the actual cycle time is hidden. This metric inversion masks the true bottleneck and prevents accurate capacity planning.",
        "confidence": 0.6,
        "improvement": 0.0,
        "consequence": 0.7,
        "evidence_sprints": [
          2
        ],
        "sprints_in_phase": 15
      },
      "AGENT-GAP-A-subrogation": {
        "phase": "applicable",
        "type": "friction",
        "description": "Agents surfaced missing handoff context at subrogation (11 decisions)",
        "hypothesis": "If we standardize the handoff into subrogation, agents will stop surfacing missing context",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.5,
        "evidence_sprints": [
          2,
          3,
          4,
          5,
          6,
          7,
          8,
          9,
          10,
          11,
          12,
          13,
          14,
          15,
          16
        ],
        "sprints_in_phase": 2
      },
      "AGENT-GAP-A-damage_estimation_human": {
        "phase": "theoretical",
        "type": "friction",
        "description": "Agents surfaced missing handoff context at damage_estimation_human (7 decisions)",
        "hypothesis": "If we standardize the handoff into damage_estimation_human, agents will stop surfacing missing context",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.36363636363636365,
        "evidence_sprints": [
          2
        ],
        "sprints_in_phase": 15
      },
      "AGENT-GAP-A-payment": {
        "phase": "theoretical",
        "type": "friction",
        "description": "Agents surfaced missing handoff context at payment (7 decisions)",
        "hypothesis": "If we standardize the handoff into payment, agents will stop surfacing missing context",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.2,
        "evidence_sprints": [
          2
        ],
        "sprints_in_phase": 15
      },
      "AGENT-GAP-A-fnol_intake": {
        "phase": "applicable",
        "type": "friction",
        "description": "Agents surfaced missing handoff context at fnol_intake (4 decisions)",
        "hypothesis": "If we standardize the handoff into fnol_intake, agents will stop surfacing missing context",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.5,
        "evidence_sprints": [
          2,
          3,
          4,
          5,
          6,
          7,
          8,
          9,
          10,
          11,
          12,
          13,
          14,
          15,
          16
        ],
        "sprints_in_phase": 13
      },
      "AGENT-GAP-A-investigation_human": {
        "phase": "applicable",
        "type": "friction",
        "description": "Agents surfaced missing handoff context at investigation_human (5 decisions)",
        "hypothesis": "If we standardize the handoff into investigation_human, agents will stop surfacing missing context",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.5,
        "evidence_sprints": [
          2,
          3,
          4,
          5,
          6,
          7,
          8,
          9,
          10,
          11,
          12,
          13,
          14,
          15,
          16
        ],
        "sprints_in_phase": 9
      },
      "AGENT-GAP-A-coverage_verification_human": {
        "phase": "applicable",
        "type": "friction",
        "description": "Agents surfaced missing handoff context at coverage_verification_human (6 decisions)",
        "hypothesis": "If we standardize the handoff into coverage_verification_human, agents will stop surfacing missing context",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.5454545454545454,
        "evidence_sprints": [
          2,
          3,
          4,
          5,
          6,
          7,
          8,
          9,
          10,
          11,
          12,
          13,
          14,
          15,
          16
        ],
        "sprints_in_phase": 13
      },
      "AGENT-GAP-A-liability_determination_human": {
        "phase": "applicable",
        "type": "friction",
        "description": "Agents surfaced missing handoff context at liability_determination_human (6 decisions)",
        "hypothesis": "If we standardize the handoff into liability_determination_human, agents will stop surfacing missing context",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.5454545454545454,
        "evidence_sprints": [
          2,
          3,
          4,
          5,
          6,
          7,
          8,
          9,
          10,
          11,
          12,
          13,
          14,
          15,
          16
        ],
        "sprints_in_phase": 13
      },
      "AGENT-GAP-A-settlement_human": {
        "phase": "replicable",
        "type": "friction",
        "description": "Agents surfaced missing handoff context at settlement_human (6 decisions)",
        "hypothesis": "If we standardize the handoff into settlement_human, agents will stop surfacing missing context",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.6875,
        "evidence_sprints": [
          2,
          3,
          4,
          5,
          6,
          7,
          8,
          9,
          10,
          11,
          12,
          13,
          14,
          15,
          16
        ],
        "sprints_in_phase": 12
      },
      "AGENT-GAP-A-approval": {
        "phase": "applicable",
        "type": "friction",
        "description": "Agents surfaced missing handoff context at approval (6 decisions)",
        "hypothesis": "If we standardize the handoff into approval, agents will stop surfacing missing context",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.5,
        "evidence_sprints": [
          2,
          3,
          4,
          5,
          6,
          7,
          8,
          9,
          10,
          11,
          12,
          13,
          14,
          15,
          16
        ],
        "sprints_in_phase": 9
      },
      "A-S03-P00": {
        "phase": "impact",
        "type": "friction",
        "description": "Translation debt has spiked dramatically to 37.97%, with a handoff failure rate of 80%. The AI pipeline is producing outputs that lose meaning when passed to human agents, particularly at the approval handoff (decision 3: 'AI processed approval but its output lost meaning at the handoff'). Diana is experiencing this directly (decisions 8, 12, 17 all show translation=True), and her exhaustion is at 8.0 with near-zero AI trust.",
        "hypothesis": "The AI pipeline is auto-processing simple claims end-to-end (coverage, settlement, approval), but when it escalates or when human agents need to take over mid-stream, the AI's internal reasoning/context is not being translated into human-understandable format. This creates a 'black box' handoff where humans must reconstruct meaning from scratch, causing rework and frustration.",
        "confidence": 0.95,
        "improvement": 0.0,
        "consequence": 0.95,
        "evidence_sprints": [
          3
        ],
        "sprints_in_phase": 11
      },
      "A-S03-P01": {
        "phase": "impact",
        "type": "friction",
        "description": "A severe trust bifurcation is emerging. Jordan (ai_trust=1.00, 75 decisions, +15/-0) and Kathryn (0.84, +5/-0) have perfect positive AI experiences, while Diana (0.00, -27 negative experiences), Tommy (0.00, -12), and Sanjay (0.13, -21) have deeply negative experiences. The negative experiences are concentrated in human-touch roles (settlement, subrogation, liability determination) while positive experiences are in simple classification tasks.",
        "hypothesis": "AI is excellent at simple, well-defined tasks (FNOL classification, coverage verification) but fails at complex judgment tasks (settlement negotiation, liability determination). Agents in complex roles are being burned by AI recommendations that don't account for nuance, while agents in simple roles see AI as a reliable assistant. This is creating a 'trust caste system' where AI trust is determined by role, not by the technology itself.",
        "confidence": 0.92,
        "improvement": 0.0,
        "consequence": 0.88,
        "evidence_sprints": [
          3
        ],
        "sprints_in_phase": 11
      },
      "A-S03-P02": {
        "phase": "theoretical",
        "type": "friction",
        "description": "Exception rate has risen to 7.59% (from 6.42% in Sprint 2), but more concerning is the pattern: exceptions are concentrated in AI pipeline handoffs (decision 3 shows exception=True at the approval step). The exception rate is not random\u2014it's systematically tied to translation failures.",
        "hypothesis": "The AI pipeline is generating exceptions not because the AI is wrong, but because the AI's output cannot be understood by downstream human systems. The exception is a symptom of translation debt, not a separate problem. As translation debt grows, exception rate will continue to climb.",
        "confidence": 0.85,
        "improvement": 0.0,
        "consequence": 0.75,
        "evidence_sprints": [
          3
        ],
        "sprints_in_phase": 14
      },
      "A-S03-P03": {
        "phase": "theoretical",
        "type": "friction",
        "description": "cycle_time is 0.0 while decision counts are extremely high (total 779 decisions across agents). This suggests the metric is not being tracked correctly, or the system is processing claims so fast that cycle time is being rounded to zero. However, the high handoff_failure_rate (80%) and low first_pass_accuracy (20%) indicate the process is NOT actually fast\u2014it's just that the metric is measuring the wrong thing.",
        "hypothesis": "The cycle_time metric is likely measuring only the AI pipeline's processing time (which is near-instant) and not including human rework time. The 80% handoff failure means most claims require multiple human interventions, but this rework is not being counted in cycle_time. The metric is giving false confidence in process speed.",
        "confidence": 0.8,
        "improvement": 0.0,
        "consequence": 0.7,
        "evidence_sprints": [
          3
        ],
        "sprints_in_phase": 14
      },
      "A-S03-P04": {
        "phase": "theoretical",
        "type": "friction",
        "description": "Several agents are using AI in ways that don't match the formal process. Alicia (ai_trust=0.70, 135 decisions, +28/-4) and Greg (0.72, 63 decisions, +18/-0) have high positive experiences and are likely using AI as a decision-support tool beyond the formal pipeline. Meanwhile, Nick (ai_trust=0.00, 101 decisions, +17/-1) has high positive experience but zero trust\u2014suggesting he's using AI but not acknowledging it, or the trust metric is not capturing his actual behavior.",
        "hypothesis": "Agents are discovering that AI is useful for certain tasks and are using it informally, even when the formal process doesn't require it. Nick's zero trust with high positive experience suggests he's using AI but doesn't want to admit it (perhaps due to peer pressure or fear of being seen as dependent on AI). This informal usage is not being tracked or optimized.",
        "confidence": 0.75,
        "improvement": 0.0,
        "consequence": 0.6,
        "evidence_sprints": [
          3
        ],
        "sprints_in_phase": 14
      },
      "A-S03-P05": {
        "phase": "impact",
        "type": "friction",
        "description": "Adoption resistance is persisting and deepening among experienced agents. Diana (0.00), Tommy (0.00), and Sanjay (0.13) have near-zero trust. Critically, these agents have high exhaustion (Diana=8.0, Sanjay=7.0, Tommy=4.0) and high stress (Sanjay=1.0, Tommy=0.85). The resistance is not irrational\u2014it's based on negative experiences (-27 for Diana, -21 for Sanjay, -12 for Tommy).",
        "hypothesis": "These agents are in roles where AI is failing them (settlement, subrogation, liability determination). Their resistance is a rational response to repeated AI failures. The organization has not addressed the root cause of these failures, so resistance is hardening into permanent distrust.",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.85,
        "evidence_sprints": [
          3
        ],
        "sprints_in_phase": 11
      },
      "A-S03-P06": {
        "phase": "theoretical",
        "type": "friction",
        "description": "Kathryn (ai_trust=0.84, stress=0.05, exhaustion=5.0) is handling escalated exceptions (decisions 4, 11) and approving them correctly. However, she's only made 12 decisions total, suggesting she's a bottleneck for escalations. The pattern from Sprint 2 persists\u2014Kathryn is the informal safety net for AI failures, but her low decision count suggests she's not being utilized efficiently.",
        "hypothesis": "Kathryn is being used as the final arbiter for AI escalations, but the escalation process is inefficient\u2014claims are being escalated to her only after multiple failed handoffs. Her low decision count suggests she's not the bottleneck; the bottleneck is the failed handoffs that prevent claims from reaching her.",
        "confidence": 0.7,
        "improvement": 0.0,
        "consequence": 0.55,
        "evidence_sprints": [
          3
        ],
        "sprints_in_phase": 14
      },
      "A-S03-P07": {
        "phase": "impact",
        "type": "friction",
        "description": "The bottleneck has migrated from human decision points to the AI pipeline's handoff points. In Sprint 2, the bottleneck was human agents. Now, the AI pipeline is auto-processing simple claims (decisions 1, 2, 6, 9, 21, 22, 23) but failing at the approval handoff (decision 3). The bottleneck is now the AI-to-human interface, not the human decision itself.",
        "hypothesis": "The AI pipeline has been optimized for speed and accuracy on simple claims, but the handoff protocol was not designed with the same rigor. The bottleneck has shifted from 'can the AI do it?' to 'can the AI communicate what it did?' This is a classic automation trap\u2014optimizing the machine while ignoring the human interface.",
        "confidence": 0.85,
        "improvement": 0.0,
        "consequence": 0.8,
        "evidence_sprints": [
          3
        ],
        "sprints_in_phase": 11
      },
      "A-S03-P08": {
        "phase": "impact",
        "type": "friction",
        "description": "first_pass_accuracy has collapsed to 20%, meaning 80% of claims require rework. This is directly correlated with the 80% handoff failure rate. The quality of the overall process is severely degraded, not because individual AI decisions are wrong (most AI decisions show correct=True), but because the process as a whole cannot deliver a correct result on the first pass.",
        "hypothesis": "The process is designed as a series of independent steps, but the handoffs between steps are failing. Even though each step is correct in isolation, the cumulative effect of handoff failures means the claim never gets processed correctly end-to-end. This is a systemic quality issue, not a component quality issue.",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.9,
        "evidence_sprints": [
          3
        ],
        "sprints_in_phase": 11
      },
      "A-S03-P09": {
        "phase": "theoretical",
        "type": "friction",
        "description": "The supplement_request_rate is 48%, meaning nearly half of all claims require additional information. This is likely a symptom of the translation debt\u2014when AI handoffs fail, downstream agents request more information to compensate for the missing context. Diana's decisions 8 and 17 show request_info actions that are directly tied to translation failures.",
        "hypothesis": "When AI handoffs lose meaning, the receiving human agent doesn't have enough context to make a decision, so they request more information from the claimant. This is a defensive behavior\u2014the agent is trying to reconstruct the missing context by asking for more data. This inflates the supplement request rate and slows down the process.",
        "confidence": 0.75,
        "improvement": 0.0,
        "consequence": 0.65,
        "evidence_sprints": [
          3
        ],
        "sprints_in_phase": 14
      },
      "AGENT-GAP-A-escalated_to_kathryn": {
        "phase": "empirical",
        "type": "friction",
        "description": "Agents surfaced missing handoff context at escalated_to_kathryn (1 decisions)",
        "hypothesis": "If we standardize the handoff into escalated_to_kathryn, agents will stop surfacing missing context",
        "confidence": 0.5,
        "improvement": 0.0,
        "consequence": 0.5,
        "evidence_sprints": [
          3,
          5,
          9
        ],
        "sprints_in_phase": 12
      },
      "A-S04-P00": {
        "phase": "impact",
        "type": "friction",
        "description": "Translation debt has dropped from the previous sprint but remains critically high at 25.33. The pattern is visible in Diana's workflow: she is making 140 decisions with 0 AI trust, and her decisions show a high rate of translation=True flags. Her settlement decisions (decisions 4, 6) show she is requesting information because required inputs are missing, which is a direct consequence of meaning being lost at handoffs. The AI pipeline itself is also experiencing translation failures \u2014 decision 22 shows the AI processed an approval but 'its output lost meaning at the handoff' and had to be escalated.",
        "hypothesis": "The translation debt is concentrated at the human-AI interface. Diana, who has zero AI trust, is likely translating AI outputs into human-readable formats manually, and this translation is lossy. The AI pipeline itself is also generating outputs that don't map cleanly to downstream human requirements, creating a bidirectional translation problem.",
        "confidence": 0.92,
        "improvement": 0.0,
        "consequence": 0.95,
        "evidence_sprints": [
          4
        ],
        "sprints_in_phase": 10
      },
      "A-S04-P01": {
        "phase": "impact",
        "type": "friction",
        "description": "A severe trust bifurcation has emerged. High-trust agents (Kathryn 0.89, Jordan 1.00, Alicia 0.87, Greg 0.88) are all experiencing low stress and low exhaustion, while low-trust agents (Diana 0.00, Tommy 0.08, Pat 0.12, Sanjay 0.14) are experiencing high stress and high exhaustion. Diana's ai_exp shows +0/-22, meaning she has had 22 negative AI experiences with zero positive ones. This is a stark contrast to Jordan's +16/-0 and Nick's +22/-0. The low-trust agents are making the majority of decisions (Diana 140, Tommy 159, Nick 131) while high-trust agents are making fewer (Kathryn 15, Jordan 100, Greg 87).",
        "hypothesis": "The trust collapse is driven by differential AI experience quality. Agents with positive AI experiences (Jordan, Nick, Kathryn) have built trust, while agents with negative experiences (Diana with 22 failures) have had their trust destroyed. The negative experiences are likely concentrated in complex claims where AI fails, while simple claims succeed \u2014 creating a skewed perception. The stress and exhaustion correlation suggests that low-trust agents are working harder to compensate for AI failures, creating a vicious cycle.",
        "confidence": 0.95,
        "improvement": 0.0,
        "consequence": 0.93,
        "evidence_sprints": [
          4
        ],
        "sprints_in_phase": 10
      },
      "A-S04-P02": {
        "phase": "theoretical",
        "type": "friction",
        "description": "Kathryn continues to serve as the escalation point for AI failures, but her workload has decreased significantly (15 decisions vs. previous sprints). However, the pattern persists: decision 23 shows Kathryn approving an escalated AI case that was correctly processed by AI but had been escalated due to translation issues. Meanwhile, Jordan (ai_trust=1.00, stress=0.00) is making 100 decisions with zero negative AI experiences, suggesting he may be operating as an informal AI champion who absorbs AI work without formal authority.",
        "hypothesis": "Kathryn's role as informal escalation handler is being formalized, but she is now being bypassed for some escalations (Jordan handles his own escalations). Jordan's perfect AI experience suggests he may be selectively choosing which claims to process with AI, avoiding complex cases that would generate negative experiences. This creates an informal shadow system where certain agents curate their AI usage.",
        "confidence": 0.78,
        "improvement": 0.0,
        "consequence": 0.72,
        "evidence_sprints": [
          4
        ],
        "sprints_in_phase": 13
      },
      "A-S04-P03": {
        "phase": "impact",
        "type": "friction",
        "description": "The bottleneck has shifted from the AI pipeline to the human settlement and investigation stages. Diana is handling settlement and investigation decisions (decisions 1, 4, 5, 6, 9, 19) with high translation debt and zero AI trust. Her settlement decisions are requesting information (decisions 4, 6) because she lacks required inputs. Meanwhile, the AI pipeline is processing simple claims efficiently (decisions 7, 11, 15, 16, 18, 20, 21) with high accuracy. The bottleneck is now in the moderate-complexity claims that require human judgment but suffer from translation debt.",
        "hypothesis": "The AI pipeline has successfully automated simple claims, but moderate-complexity claims require human intervention. These claims are being routed to Diana, who lacks AI support and must manually translate AI outputs. The translation debt at the human stage creates a bottleneck where Diana cannot process claims efficiently because she doesn't trust or receive properly formatted AI outputs.",
        "confidence": 0.88,
        "improvement": 0.0,
        "consequence": 0.85,
        "evidence_sprints": [
          4
        ],
        "sprints_in_phase": 10
      },
      "A-S04-P04": {
        "phase": "impact",
        "type": "friction",
        "description": "Adoption resistance has evolved from a general pattern to a specific cluster. Diana (ai_trust=0.00, 140 decisions, 22 negative AI experiences) and Tommy (ai_trust=0.08, 159 decisions, 9 negative experiences) are the primary resisters. However, Nick presents a paradox: ai_trust=0.00 but ai_exp=+22/-0 \u2014 he has had 22 positive AI experiences yet still doesn't trust AI. This suggests his resistance is not experience-based but ideological or process-based.",
        "hypothesis": "Nick's resistance is particularly concerning because he has only positive AI experiences yet still doesn't trust AI. This suggests his resistance is based on job security concerns or a philosophical objection to AI in claims processing. His high exhaustion (9.0) despite low stress (0.00) suggests he is working hard to maintain manual processes despite AI being available and successful.",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.82,
        "evidence_sprints": [
          4
        ],
        "sprints_in_phase": 10
      },
      "A-S04-P05": {
        "phase": "impact",
        "type": "friction",
        "description": "The supplement_request_rate has dropped from 48% to 36%, but remains elevated. The pattern is visible in Sanjay's approval decisions \u2014 decision 10 shows him requesting information because the claim 'has been held four times for missing fundamental documentation.' This suggests a systemic issue where claims are being passed between stages without complete information, creating a loop of requests and resubmissions. The 60% handoff_failure_rate compounds this by ensuring that even when information is provided, it may be lost in translation.",
        "hypothesis": "The coordination drag is caused by a lack of standardized information requirements across stages. Each stage has different requirements, and when claims move between stages, information that was sufficient for one stage is insufficient for another. The translation debt exacerbates this by losing information during handoffs, forcing downstream stages to request information that was already provided upstream.",
        "confidence": 0.84,
        "improvement": 0.0,
        "consequence": 0.8,
        "evidence_sprints": [
          4
        ],
        "sprints_in_phase": 10
      },
      "A-S04-P06": {
        "phase": "impact",
        "type": "friction",
        "description": "First_pass_accuracy has improved from 20% to 40%, but remains critically low. The pattern is visible in the decision log: AI pipeline decisions are consistently correct (decisions 7, 11, 15, 16, 18, 20, 21, 24 all show correct=True), but human decisions show correct=None because they are not being evaluated. This creates a blind spot \u2014 we know AI is performing well, but we have no data on human decision quality. The 60% handoff_failure_rate suggests that even when decisions are made correctly, they may not be transmitted correctly.",
        "hypothesis": "The quality debt is partially a measurement problem \u2014 human decisions are not being evaluated for correctness, so we cannot identify where quality issues originate. The 40% first_pass_accuracy may be artificially low because it only counts AI decisions, or it may be accurate but we can't tell. The handoff_failure_rate suggests that even correct decisions are being lost or corrupted during transmission.",
        "confidence": 0.87,
        "improvement": 0.0,
        "consequence": 0.88,
        "evidence_sprints": [
          4
        ],
        "sprints_in_phase": 10
      },
      "A-S04-P07": {
        "phase": "theoretical",
        "type": "friction",
        "description": "Evidence of shadow AI usage is emerging. Jordan (ai_trust=1.00, 100 decisions, +16/-0) and Nick (ai_trust=0.00, 131 decisions, +22/-0) both have high decision counts with perfect or near-perfect AI experience. However, their trust levels are diametrically opposed. This suggests that Jordan may be using AI informally without formal authorization, while Nick may be using AI but not acknowledging it. The AI Pipeline decisions (7, 11, 15, 16, 18, 20, 21) show AI being used for simple claims, but the human agents with high decision counts may be using AI tools outside the formal pipeline.",
        "hypothesis": "Some agents are using AI tools outside the formal pipeline, either because the formal pipeline is too restrictive or because they have found more efficient ways to use AI. Jordan's perfect AI experience suggests he has developed effective informal AI usage patterns. Nick's resistance despite positive AI experience may be because he's using AI informally but doesn't want to acknowledge it due to job security concerns.",
        "confidence": 0.65,
        "improvement": 0.0,
        "consequence": 0.7,
        "evidence_sprints": [
          4
        ],
        "sprints_in_phase": 13
      },
      "A-S04-P08": {
        "phase": "theoretical",
        "type": "friction",
        "description": "Exception rate has dropped from previous sprints to 5.33, which is a significant improvement. However, the exceptions that do occur are concentrated in the AI pipeline handoff (decision 22) and complex claims (decision 0). The exception at decision 22 is particularly concerning because it shows the AI pipeline itself generating an exception due to translation failure, not a processing failure. This suggests that the AI pipeline is becoming a source of exceptions rather than a solution to them.",
        "hypothesis": "The exception rate has improved because the AI pipeline is successfully handling simple claims. However, the remaining exceptions are now concentrated in the AI pipeline's own handoff failures and in complex claims that require human judgment. The AI pipeline is creating exceptions through translation failures, which is a new source of exceptions that didn't exist before AI implementation.",
        "confidence": 0.75,
        "improvement": 0.0,
        "consequence": 0.68,
        "evidence_sprints": [
          4
        ],
        "sprints_in_phase": 13
      },
      "A-S04-P09": {
        "phase": "theoretical",
        "type": "friction",
        "description": "The cycle_time metric is 0.0, which is suspicious. In a system with 60% handoff_failure_rate and 36% supplement_request_rate, cycle time should be positive. A cycle time of 0.0 suggests either the metric is being measured incorrectly, or claims are being processed in parallel rather than sequentially, or the metric is being gamed. The cost_per_claim of $359.48 with a 40% first_pass_accuracy suggests that rework costs are being hidden in the cycle time metric.",
        "hypothesis": "The cycle_time metric may be measuring only the time from first submission to final approval, ignoring the time spent in request_info loops and rework. Alternatively, the metric may be calculated based on a sample that excludes complex claims. The 0.0 value is inconsistent with the other metrics, suggesting a measurement or calculation error.",
        "confidence": 0.6,
        "improvement": 0.0,
        "consequence": 0.75,
        "evidence_sprints": [
          4
        ],
        "sprints_in_phase": 13
      },
      "A-S04-P10": {
        "phase": "impact",
        "type": "friction",
        "description": "A clear capability gap exists between agents who can effectively use AI and those who cannot. High-trust agents (Kathryn 0.89, Jordan 1.00, Alicia 0.87, Greg 0.88) have low stress and exhaustion, while low-trust agents (Diana 0.00, Pat 0.12, Sanjay 0.14) have high stress and exhaustion. The gap is particularly visible in decision quality: agents with high AI trust are making decisions with AI assistance (decision 23, 24), while low-trust agents are making decisions without AI (decisions 1-6, 9, 10, 13, 14, 19). This creates a two-tier system where some agents have access to AI capabilities and others don't.",
        "hypothesis": "The capability gap is driven by differential access to AI tools and training. High-trust agents have likely received better training or have more experience with AI systems. Low-trust agents may have been given AI tools without adequate support, leading to negative experiences and capability gaps. The stress and exhaustion differences suggest that low-trust agents are working harder to compensate for their lack of AI capability.",
        "confidence": 0.85,
        "improvement": 0.0,
        "consequence": 0.83,
        "evidence_sprints": [
          4
        ],
        "sprints_in_phase": 10
      },
      "A-S04-P12": {
        "phase": "theoretical",
        "type": "friction",
        "description": "Authority ambiguity is evident in the escalation patterns. Jordan (ai_trust=1.00) is making escalation decisions (decision 0) that should go to Kathryn (the formal escalation handler). Meanwhile, Kathryn is approving escalated cases (decision 23) that were escalated by the AI pipeline. The decision log shows no clear pattern of who has authority to escalate, approve, or reject at different stages. This ambiguity is likely contributing to the 60% handoff_failure_rate.",
        "hypothesis": "The AI transformation has blurred traditional authority boundaries. Agents who trust AI (like Jordan) are taking on escalation authority that was previously reserved for supervisors. The AI pipeline itself is making escalation decisions, creating a new authority layer that didn't exist before. This ambiguity creates confusion about who is responsible for what, leading to handoff failures.",
        "confidence": 0.7,
        "improvement": 0.0,
        "consequence": 0.77,
        "evidence_sprints": [
          4
        ],
        "sprints_in_phase": 13
      },
      "A-S04-P13": {
        "phase": "theoretical",
        "type": "friction",
        "description": "Agents are engaging in boundary work to protect their professional autonomy from AI encroachment. Diana (ai_trust=0.00, 140 decisions) is making the most decisions while completely rejecting AI, suggesting she is actively working to maintain her professional judgment as the primary decision-maker. Tommy (ai_trust=0.08, 159 decisions) shows a similar pattern. These agents are likely using their high decision counts to demonstrate their value and necessity, potentially as a defensive response to AI implementation.",
        "hypothesis": "Agents who feel threatened by AI are engaging in boundary work by maximizing their decision-making activity. This serves two purposes: it demonstrates their continued value to the organization, and it limits AI's role by keeping decisions in human hands. The high decision counts may be a form of resistance through productivity.",
        "confidence": 0.68,
        "improvement": 0.0,
        "consequence": 0.66,
        "evidence_sprints": [
          4
        ],
        "sprints_in_phase": 13
      },
      "A-S04-P14": {
        "phase": "impact",
        "type": "friction",
        "description": "A data seam is visible between the AI pipeline and human decision-makers. The AI pipeline produces structured, standardized outputs (decisions 7, 11, 15, 16, 18, 20, 21) that are consistently correct. However, when these outputs need to be integrated with human decision-making, the seam becomes visible \u2014 decision 22 shows the AI pipeline's output losing meaning at the handoff. Human agents like Diana are receiving AI outputs that don't integrate cleanly with their workflow, creating a seam where data quality degrades.",
        "hypothesis": "The AI pipeline and human workflow operate on different data models. AI produces structured data that doesn't map cleanly to the human decision-making process. This creates a seam where data must be translated, and translation is lossy. The seam is particularly problematic for complex claims that require human judgment.",
        "confidence": 0.8,
        "improvement": 0.0,
        "consequence": 0.86,
        "evidence_sprints": [
          4
        ],
        "sprints_in_phase": 10
      },
      "A-S04-P15": {
        "phase": "replicable",
        "type": "friction",
        "description": "Early signs of an attrition spiral are visible. Diana (exhaustion=8.0, stress=0.15, ai_trust=0.00) and Nick (exhaustion=9.0, stress=0.00, ai_trust=0.00) have the highest exhaustion levels in the system. Both are making high numbers of decisions (140 and 131 respectively) while completely rejecting AI. This combination of high workload, high exhaustion, and zero AI trust is a recipe for burnout and attrition. If these agents leave, their workload would be redistributed to other agents, potentially creating a cascade.",
        "hypothesis": "Agents who reject AI are forced to work harder to compensate for the lack of AI assistance. This increases their exhaustion. As exhaustion increases, they become more resistant to AI (because they don't have the energy to learn new tools), creating a vicious cycle. If they leave, their workload would be redistributed, potentially overwhelming other agents.",
        "confidence": 0.72,
        "improvement": 0.0,
        "consequence": 0.9,
        "evidence_sprints": [
          4
        ],
        "sprints_in_phase": 11
      },
      "A-S04-P17": {
        "phase": "theoretical",
        "type": "friction",
        "description": "A positive trust cascade is emerging among high-trust agents. Kathryn (0.89), Jordan (1.00), Alicia (0.87), Greg (0.88), and Rachel (0.65) are all showing high trust with low stress and exhaustion. These agents are likely influencing each other positively, creating a virtuous cycle where successful AI usage reinforces trust. Decision 23 shows Kathryn approving an AI-processed case, and decision 24 shows Tommy using AI successfully. This positive cascade could be leveraged to influence low-trust agents.",
        "hypothesis": "High-trust agents are experiencing positive AI outcomes, which reinforces their trust. Their low stress and exhaustion allow them to use AI more effectively, creating a positive feedback loop. These agents may be informally sharing their positive experiences with each other, amplifying the cascade.",
        "confidence": 0.76,
        "improvement": 0.0,
        "consequence": 0.64,
        "evidence_sprints": [
          4
        ],
        "sprints_in_phase": 13
      },
      "A-S05-P00": {
        "phase": "impact",
        "type": "friction",
        "description": "Translation debt is critically high (27.27) and is now the primary driver of workflow failure. The AI Pipeline is producing outputs that lose meaning at handoffs, causing downstream agents to reject or escalate work. The handoff_failure_rate has spiked to 60%, and the exception_rate is 5.19%. The AI Pipeline's approval step is repeatedly flagged with 'output lost meaning at the handoff \u2014 downstream mu...' indicating a systemic semantic breakdown between AI-generated outputs and human-readable context.",
        "hypothesis": "The AI Pipeline is generating structured outputs (e.g., approval decisions) that lack the narrative context or justification that human reviewers need to validate the decision. The pipeline is optimized for speed and accuracy on simple claims, but its output format is not compatible with the human review process, creating a semantic gap that forces downstream agents to escalate or request more information.",
        "confidence": 0.95,
        "improvement": 0.0,
        "consequence": 0.95,
        "evidence_sprints": [
          5
        ],
        "sprints_in_phase": 9
      },
      "A-S05-P01": {
        "phase": "impact",
        "type": "friction",
        "description": "The attrition spiral is accelerating. Diana (exhaustion=8.0, stress=0.05, ai_trust=0.00) and Nick (exhaustion=9.0, stress=0.00, ai_trust=0.00) are showing critical exhaustion levels. Diana has made 176 decisions, and Nick has made 163, both with zero AI trust. Their stress levels are paradoxically low, suggesting emotional detachment or resignation. Pat (stress=0.95, exhaustion=5.0, ai_trust=0.02) and Sanjay (stress=0.95, exhaustion=7.0, ai_trust=0.00) are showing high stress with negative AI experiences (Pat: +14/-12, Sanjay: +24/-5).",
        "hypothesis": "These agents are being overwhelmed by the volume of decisions and the need to compensate for AI failures. Diana's repeated handling of the same claims (Decision 6: 'I've been handed this claim three times now') indicates she is doing rework caused by translation debt. The low stress with high exhaustion suggests they have moved past active stress into burnout, which is more dangerous as they may disengage or leave.",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.9,
        "evidence_sprints": [
          5
        ],
        "sprints_in_phase": 9
      },
      "A-S05-P02": {
        "phase": "impact",
        "type": "friction",
        "description": "A trust collapse is emerging among high-exhaustion, high-decision agents. While there is a positive trust cascade among low-exhaustion agents (Jordan, Alicia, Greg, Kathryn), the agents with the most decision experience (Diana, Nick, Tommy, Sanjay) have zero or near-zero AI trust. Tommy (decisions=198, ai_trust=0.00, ai_exp=+17/-8) has the most decisions but zero trust, indicating that experience with AI is not building trust \u2014 it's destroying it. The negative experiences are concentrated among those who see the failures firsthand.",
        "hypothesis": "The agents with zero trust are those who have experienced AI failures directly (Diana: -25 negative experiences, Tommy: -8, Sanjay: -5). Nick has +20 positive and 0 negative but still has zero trust, suggesting he may be observing the failures of others or the systemic issues (translation debt) rather than his own experiences. The high-trust agents (Jordan, Alicia) may be processing simpler claims where AI succeeds, creating a split experience.",
        "confidence": 0.85,
        "improvement": 0.0,
        "consequence": 0.8,
        "evidence_sprints": [
          5
        ],
        "sprints_in_phase": 9
      },
      "A-S05-P03": {
        "phase": "theoretical",
        "type": "friction",
        "description": "The bottleneck has migrated from the AI Pipeline to the human approval and settlement steps. The AI Pipeline is auto-processing simple claims successfully (Decisions 1-3, 11, 14-15, 20-21), but human steps like approval (Sanjay, Decision 5) and settlement_human (Diana, Decisions 6, 8) are becoming choke points. Sanjay (stress=0.95) and Diana (exhaustion=8.0) are the primary bottleneck agents, and their decisions are frequently 'request_info' or 'escalate' rather than 'approve'.",
        "hypothesis": "The AI Pipeline is handling simple claims efficiently, but the remaining claims reaching humans are complex or have translation issues. The humans are being asked to make decisions on incomplete or ambiguous information (due to translation debt), forcing them to request more info or escalate. This creates a bottleneck at the human review stage.",
        "confidence": 0.85,
        "improvement": 0.0,
        "consequence": 0.75,
        "evidence_sprints": [
          5
        ],
        "sprints_in_phase": 12
      },
      "A-S05-P04": {
        "phase": "theoretical",
        "type": "friction",
        "description": "There is evidence of informal control exposure where agents are making decisions outside the formal AI-driven process. Tommy (Decision 10) is making subrogation decisions that 'are not settlement approval' but still cannot proceed, indicating he is being pulled into decisions outside his formal role. Diana (Decision 19) is approving investigation steps, which may be outside her formal authority. The high number of human decisions (Diana: 176, Tommy: 198, Alicia: 230) suggests agents are compensating for AI gaps by taking on additional informal responsibilities.",
        "hypothesis": "The formal process is not handling the full scope of work, so agents are stepping in to fill gaps. Tommy is making subrogation decisions that may not be his formal responsibility, and Diana is approving investigations. This informal control is necessary to keep the workflow moving but creates risk of decisions being made without proper authority or oversight.",
        "confidence": 0.7,
        "improvement": 0.0,
        "consequence": 0.7,
        "evidence_sprints": [
          5
        ],
        "sprints_in_phase": 12
      },
      "A-S05-P05": {
        "phase": "impact",
        "type": "friction",
        "description": "Exception rate is at 5.19%, and the exceptions are concentrated in the AI Pipeline's approval step and Diana's settlement_human step. The AI Pipeline exceptions (Decisions 9, 18, 22) are all 'translation=True, exception=True', indicating that the AI is generating exceptions because its output cannot be understood downstream. Diana's exceptions (Decision 8) are also translation-related. This suggests that exceptions are not due to claim complexity but due to system communication failures.",
        "hypothesis": "The AI Pipeline is generating exceptions because it cannot translate its output into a format that downstream systems or humans can use. This is a technical failure, not a business rule failure. The exceptions are being triggered by the system's inability to communicate, not by the claim's characteristics.",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.85,
        "evidence_sprints": [
          5
        ],
        "sprints_in_phase": 9
      },
      "A-S05-P06": {
        "phase": "theoretical",
        "type": "friction",
        "description": "There is evidence of shadow AI usage, where agents are using AI recommendations but not formally acknowledging it. Tommy (Decision 13) used AI and it was correct, but his justification reads like a human decision ('This is a simple auto claim with clear liability...'). Jordan (Decision 0) explicitly states 'The AI recommendation aligns with the standard path,' but other agents like Tommy may be silently using AI without documenting it. The high number of decisions with 'AI used=True' but human-style justifications suggests agents are blending AI input with their own reasoning, potentially hiding AI reliance.",
        "hypothesis": "Agents may be using AI recommendations but presenting them as their own decisions to avoid accountability or because they don't fully trust the AI but find it useful. This creates a hidden dependency on AI that is not formally tracked, making it difficult to assess the true impact of AI on decision quality.",
        "confidence": 0.6,
        "improvement": 0.0,
        "consequence": 0.5,
        "evidence_sprints": [
          5
        ],
        "sprints_in_phase": 12
      },
      "A-S05-P07": {
        "phase": "theoretical",
        "type": "friction",
        "description": "A pidgin language is emerging between the AI Pipeline and human agents. The AI Pipeline produces outputs that are partially understood but require human interpretation. Diana's repeated requests for 'the same critical gap' (Decision 6) and Sanjay's comment about claims being 'passed through five steps, each approving it with the same c...' (Decision 17) suggest that humans are developing a shared understanding of what the AI means, but this understanding is incomplete and leads to rework.",
        "hypothesis": "The AI Pipeline's output format is not fully compatible with human workflows, so humans are developing their own interpretations of what the AI means. This pidgin is inefficient and error-prone, as different agents may interpret the AI output differently.",
        "confidence": 0.75,
        "improvement": 0.0,
        "consequence": 0.7,
        "evidence_sprints": [
          5
        ],
        "sprints_in_phase": 12
      },
      "A-S05-P08": {
        "phase": "impact",
        "type": "friction",
        "description": "There is a capability gap between what the AI Pipeline can handle and what the organization needs. The AI Pipeline is successfully processing simple claims (auto, no injuries, modest amounts), but the organization is also receiving moderate-complexity claims (disputed liability, higher amounts) that the AI cannot handle. Diana (Decision 16) is handling a 'moderate-complexity claim with disputed liability' that requires human judgment. The AI Pipeline is not being used for these claims, but the human agents are overwhelmed by them.",
        "hypothesis": "The AI Pipeline was designed for simple claims, but the organization's claim portfolio includes more complex cases. The AI cannot handle these, so they fall to human agents, who are already overloaded with rework from translation debt.",
        "confidence": 0.8,
        "improvement": 0.0,
        "consequence": 0.8,
        "evidence_sprints": [
          5
        ],
        "sprints_in_phase": 9
      },
      "A-S05-P09": {
        "phase": "impact",
        "type": "friction",
        "description": "The cycle_time metric is 0.0, which is suspiciously low and likely indicates a measurement failure or metric inversion. Given the high handoff_failure_rate (60%) and the number of claims being sent back for more information, cycle time should be increasing. The 0.0 value suggests that either the metric is not being tracked correctly, or the system is measuring only the AI processing time and not the full end-to-end cycle including human rework.",
        "hypothesis": "The cycle_time metric is likely only measuring the AI Pipeline's processing time, not the time spent in human review, rework, or waiting. This gives a false sense of efficiency and masks the true cost of the translation debt.",
        "confidence": 0.85,
        "improvement": 0.0,
        "consequence": 0.8,
        "evidence_sprints": [
          5
        ],
        "sprints_in_phase": 9
      },
      "A-S05-P10": {
        "phase": "theoretical",
        "type": "friction",
        "description": "There is significant coordination drag between the AI Pipeline and human agents, and between human agents themselves. The AI Pipeline is auto-processing claims, but when it hands off to humans, the humans don't have the context they need. Diana (Decision 6) has been handed the same claim three times, indicating that coordination between steps is failing. The high number of 'request_info' decisions (Sanjay, Diana, Tommy) suggests that agents are spending time requesting information that should have been provided earlier in the process.",
        "hypothesis": "The handoff process between steps is not well-defined. Information that is critical for decision-making is not being passed along, forcing agents to request it repeatedly. This is a coordination failure, not a capability failure.",
        "confidence": 0.8,
        "improvement": 0.0,
        "consequence": 0.7,
        "evidence_sprints": [
          5
        ],
        "sprints_in_phase": 12
      },
      "A-S06-P00": {
        "phase": "impact",
        "type": "friction",
        "description": "The AI Pipeline is producing outputs that lose meaning at the handoff to downstream human or system processes. This is visible in the settlement_ai escalations where the AI's output 'lost meaning at the handoff' and triggered exceptions. The translation_debt_index is 21.57, and the exception_rate is 5.88%, but the handoff_failure_rate is a striking 64%.",
        "hypothesis": "The AI Pipeline is generating outputs in a format or with a semantic structure that downstream consumers (human or system) cannot interpret. This is not a simple data format issue but a deeper semantic mismatch \u2014 the AI's internal representation of 'settlement approved' does not map to what the next step expects. The 64% handoff failure rate suggests this is systemic, not isolated to complex claims.",
        "confidence": 0.92,
        "improvement": 0.0,
        "consequence": 0.95,
        "evidence_sprints": [
          6,
          7,
          10,
          11,
          13,
          14,
          16
        ],
        "sprints_in_phase": 8
      },
      "A-S06-P01": {
        "phase": "impact",
        "type": "friction",
        "description": "There is a severe bifurcation in AI trust. High-volume decision-makers (Diana, Tommy, Nick, Sanjay, Pat) show near-zero trust (0.00-0.03), while lower-volume or specialized agents (Jordan, Greg, Alicia, Kathryn) show high trust (0.82-1.00). Diana, who has made 215 decisions, has 0.00 trust and 28 negative AI experiences. This is not random \u2014 it correlates with decision volume and negative exposure.",
        "hypothesis": "Agents who are forced to interact with AI outputs that are frequently wrong or meaningless (like Diana's 28 negative experiences) develop a learned helplessness and reject AI entirely. Agents who only see AI succeed (Jordan, Greg) develop unconditional trust. The pattern is not about the AI's actual accuracy but about the distribution of failures \u2014 those who see failures up close lose trust completely, while those who don't see failures become overconfident.",
        "confidence": 0.88,
        "improvement": 0.0,
        "consequence": 0.9,
        "evidence_sprints": [
          6
        ],
        "sprints_in_phase": 8
      },
      "A-S06-P02": {
        "phase": "impact",
        "type": "friction",
        "description": "The AI Pipeline can only handle simple claims reliably. The data shows AI auto-processing succeeding on simple claims (e.g., $1,752.97, $597.57, $6,338.51) but failing on settlement for more complex cases. Meanwhile, human agents are being pulled into complex claims that require judgment, but they are also handling simple claims that the AI could do, creating inefficiency. The first_pass_accuracy of 36% suggests the overall system is failing to get claims right the first time.",
        "hypothesis": "The AI Pipeline was trained or designed primarily on simple, low-variance claims. When it encounters complex claims (multi-vehicle, disputed liability, high value), it either produces outputs that don't translate or makes errors. The organization has not yet defined a clear boundary for what the AI can handle, so it attempts everything and fails on the complex end, while humans are still doing simple claims that the AI could handle, wasting their capacity.",
        "confidence": 0.85,
        "improvement": 0.0,
        "consequence": 0.88,
        "evidence_sprints": [
          6
        ],
        "sprints_in_phase": 8
      },
      "A-S06-P03": {
        "phase": "impact",
        "type": "friction",
        "description": "The cycle_time metric is 0.0, which is physically impossible for any real process. This indicates either a measurement failure or a definitional problem where the metric is not capturing actual elapsed time. Combined with the 64% handoff failure rate, this suggests the organization is flying blind on its most important efficiency metric.",
        "hypothesis": "The cycle_time metric is likely being measured from the moment a claim enters the AI Pipeline to the moment the AI produces an output, but not including the time spent in human queues, exception handling, or rework. Since the AI processes in milliseconds, the metric reads 0.0. This hides the true end-to-end time, which is likely much longer given the 64% handoff failure rate and the need for human intervention.",
        "confidence": 0.95,
        "improvement": 0.0,
        "consequence": 0.9,
        "evidence_sprints": [
          6
        ],
        "sprints_in_phase": 8
      },
      "A-S06-P04": {
        "phase": "impact",
        "type": "friction",
        "description": "There is significant friction between the AI Pipeline and human agents, and between human agents themselves. Diana is making 215 decisions with high exhaustion (8.0), while other agents like Tricia and Ron have made 0 decisions. The workload is extremely uneven, and the handoff failures are forcing Diana to redo work that should have been completed by the AI.",
        "hypothesis": "The AI Pipeline is routing complex claims to a small set of 'trusted' human agents (Diana, Tommy) while other agents are underutilized. The handoff failures from the AI create additional rework for these same agents, compounding their workload. The organization has not implemented a load-balancing mechanism that accounts for AI failures, so the burden falls on the most capable or most available agents, creating a bottleneck.",
        "confidence": 0.82,
        "improvement": 0.0,
        "consequence": 0.85,
        "evidence_sprints": [
          6
        ],
        "sprints_in_phase": 8
      },
      "A-S06-P05": {
        "phase": "theoretical",
        "type": "friction",
        "description": "There is evidence that agents are using AI outputs informally without formally acknowledging it. Pat has 14 positive AI experiences and 0 negative, but his ai_trust is only 0.02. Sanjay has 23 positive and 5 negative experiences but trust is 0.03. This suggests they are using AI but not trusting it formally, possibly because they are overriding or re-doing AI work without recording it as AI-assisted.",
        "hypothesis": "Agents like Pat and Sanjay are looking at AI outputs to inform their decisions but are not formally accepting or rejecting the AI recommendation in the system. They may be doing this to avoid accountability if the AI is wrong, or because the formal AI acceptance process is cumbersome. This creates a hidden dependency on AI that is not tracked, making it impossible to measure true AI effectiveness.",
        "confidence": 0.78,
        "improvement": 0.0,
        "consequence": 0.7,
        "evidence_sprints": [
          6
        ],
        "sprints_in_phase": 11
      },
      "A-S06-P06": {
        "phase": "impact",
        "type": "friction",
        "description": "The exception_rate is 5.88%, but the handoff_failure_rate is 64%. This suggests that many handoff failures are NOT being recorded as exceptions. The system is undercounting exceptions because the AI Pipeline's failures are being silently absorbed by human agents who fix the issues without formally escalating. This masks the true severity of the AI's translation problem.",
        "hypothesis": "When the AI Pipeline produces a garbled output, the downstream human agent often recognizes the issue and fixes it silently rather than formally escalating it as an exception. This is because the exception process is time-consuming and the agent knows the fix. The organization's exception_rate therefore only captures the most severe failures, not the pervasive translation issues.",
        "confidence": 0.86,
        "improvement": 0.0,
        "consequence": 0.92,
        "evidence_sprints": [
          6
        ],
        "sprints_in_phase": 8
      },
      "A-S06-P07": {
        "phase": "replicable",
        "type": "friction",
        "description": "Several agents with zero decisions (Ron, Tricia, Mike, Leslie) have moderate to high stress levels (0.60-0.80) and low-to-moderate AI trust (0.35-0.77). These agents are not engaging with the system at all, yet they are stressed. This suggests they are being bypassed by the workflow and are either anxious about being replaced or frustrated by being underutilized.",
        "hypothesis": "The AI Pipeline is handling more simple claims, reducing the need for these agents' involvement. They are being kept on the payroll but given no work, creating anxiety about job security. Their stress is a response to perceived obsolescence. They have not been retrained for higher-complexity claims or given new responsibilities.",
        "confidence": 0.75,
        "improvement": 0.0,
        "consequence": 0.8,
        "evidence_sprints": [
          6
        ],
        "sprints_in_phase": 9
      },
      "A-S06-P08": {
        "phase": "impact",
        "type": "friction",
        "description": "The bottleneck has shifted from the AI Pipeline (which processes simple claims quickly) to the human agents handling complex claims, specifically Diana and Tommy. Diana has 215 decisions and exhaustion 8.0, Tommy has 236 decisions. The AI is not the bottleneck for simple claims, but the human pipeline is now the critical path for all complex claims, and it is overloaded.",
        "hypothesis": "As the AI Pipeline became more capable on simple claims, the volume of complex claims reaching humans did not decrease, but the human capacity was not increased. The AI's success on simple claims freed up some human time, but the handoff failures and translation issues created new work that offset those gains. The net effect is that the human bottleneck is now more severe than before, and it is concentrated on a few individuals.",
        "confidence": 0.84,
        "improvement": 0.0,
        "consequence": 0.9,
        "evidence_sprints": [
          6
        ],
        "sprints_in_phase": 8
      },
      "A-S06-P09": {
        "phase": "impact",
        "type": "friction",
        "description": "A pidgin language is emerging between the AI Pipeline and human agents. The AI produces outputs that are 'close enough' to meaningful but not quite right, and humans are developing workarounds to interpret them. Diana's translation=True flags on decisions 7, 9, 11, and 22 suggest she is spending significant effort translating AI outputs into actionable information. This is not a one-off issue but a systematic pattern of meaning distortion.",
        "hypothesis": "The AI Pipeline and human agents have developed different internal representations of claims. The AI uses structured data and confidence scores, while humans use narrative context and judgment. When the AI outputs a decision, it does not include the reasoning in a way humans can easily consume, so humans must 'translate' the AI's output into their own mental model. This translation is error-prone and time-consuming.",
        "confidence": 0.8,
        "improvement": 0.0,
        "consequence": 0.85,
        "evidence_sprints": [
          6
        ],
        "sprints_in_phase": 8
      },
      "A-S07-P01": {
        "phase": "impact",
        "type": "friction",
        "description": "The handoff_failure_rate has reached 68% (up from 64% in Sprint 6), while exception_rate is 6.37% and supplement_request_rate is 16%. The decision log shows a pattern where Tommy (subrogation specialist) makes multiple sequential decisions on the same claim (CL-07-0009) - decisions 4, 6, 7, 8 - with escalating actions (accept_ai, approve, escalate, approve). This suggests handoffs between stages are failing, causing work to bounce back to the same agent repeatedly.",
        "hypothesis": "The handoff protocol between stages is failing because downstream stages are rejecting work that upstream stages consider complete. This is likely due to misaligned criteria - the AI Pipeline and early human stages approve based on simple-claim criteria, but later stages (like subrogation) require additional information or different evaluation standards. The 68% failure rate suggests a systemic mismatch in what constitutes 'ready for next stage'.",
        "confidence": 0.88,
        "improvement": 0.0,
        "consequence": 0.9,
        "evidence_sprints": [
          7
        ],
        "sprints_in_phase": 7
      },
      "A-S07-P02": {
        "phase": "impact",
        "type": "friction",
        "description": "A clear bifurcation in AI trust is emerging. High-trust agents (Jordan: 1.00, Kathryn: 0.83, Tricia: 0.78, Alicia: 0.79, Greg: 0.74) have low stress and high decision counts. Low-trust agents (Diana: 0.00, Tommy: 0.00, Nick: 0.00, Pat: 0.03, Sanjay: 0.04) have high stress and high decision counts. Notably, Diana has 260 decisions with 31 negative AI experiences and 0 trust, while Nick has 221 decisions with 22 positive experiences but still 0 trust.",
        "hypothesis": "Trust is not solely determined by AI accuracy - it's influenced by the type of decisions agents make. Diana (Senior Claims Adjuster) handles complex, high-stakes claims where AI errors are more consequential. Nick's 0 trust despite 22 positive experiences suggests he may have had one early negative experience that shaped his perception, or he's in a role where AI recommendations are less useful. The low-trust agents are also the ones with highest stress, suggesting a feedback loop where distrust increases cognitive load.",
        "confidence": 0.85,
        "improvement": 0.0,
        "consequence": 0.85,
        "evidence_sprints": [
          7
        ],
        "sprints_in_phase": 7
      },
      "A-S07-P03": {
        "phase": "impact",
        "type": "friction",
        "description": "Four agents (Ron, Tricia, Mike, Leslie) have zero decisions this sprint, continuing a pattern from Sprint 6. These agents have moderate to high stress (0.60-0.80) and varying AI trust levels (0.36-0.78). Notably, Tricia has high AI trust (0.78) but zero decisions, suggesting she's willing to use AI but isn't being assigned work. Mike has 0.65 trust but 8.0 exhaustion, indicating he may be burned out from previous sprints.",
        "hypothesis": "These agents are being systematically excluded from the workflow, likely because the AI Pipeline is auto-processing simple claims that would normally go to them. The remaining complex claims are being routed to specialists like Diana and Tommy. This creates a paradox where agents who are ready to use AI (Tricia, Mike) are idle, while agents who distrust AI (Diana, Sanjay) are overloaded.",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.8,
        "evidence_sprints": [
          7
        ],
        "sprints_in_phase": 7
      },
      "A-S07-P04": {
        "phase": "theoretical",
        "type": "friction",
        "description": "Exception rate is 6.37%, up from 5.88% in Sprint 6. The decision log shows Tommy escalating a claim (decision 7) that he had previously approved (decision 6), indicating that exceptions are not just about claim complexity but about process failures. The escalation happened after an initial approval, suggesting the exception was triggered by downstream feedback rather than initial assessment.",
        "hypothesis": "The exception rate is being driven by process inconsistencies rather than genuine claim complexity. Tommy's sequence (approve \u2192 escalate \u2192 approve) suggests that the escalation was triggered by an external factor (perhaps a handoff failure or a system flag) rather than a change in claim circumstances. This indicates that exceptions are being used as a workaround for broken handoffs.",
        "confidence": 0.78,
        "improvement": 0.0,
        "consequence": 0.75,
        "evidence_sprints": [
          7
        ],
        "sprints_in_phase": 10
      },
      "A-S07-P05": {
        "phase": "impact",
        "type": "friction",
        "description": "The cycle_time metric is 0.0, which is mathematically impossible for a system processing claims. This suggests the metric is either not being tracked correctly or is being gamed. Combined with cost_per_claim at $400.08 (which seems high for simple claims), this indicates that the metrics being reported do not reflect actual system performance.",
        "hypothesis": "The cycle_time metric is likely being measured only for AI Pipeline steps (which are instantaneous) and not for human steps. This creates a false impression of efficiency while hiding the true end-to-end time. The cost_per_claim of $400 suggests that despite AI automation, the cost is not decreasing, possibly because human rework is expensive.",
        "confidence": 0.82,
        "improvement": 0.0,
        "consequence": 0.88,
        "evidence_sprints": [
          7
        ],
        "sprints_in_phase": 7
      },
      "A-S07-P06": {
        "phase": "applicable",
        "type": "friction",
        "description": "The shadow_ai pattern from Sprint 6 persists. Pat (ai_trust=0.03) has 78 decisions with 11 positive and 1 negative AI experience, yet still distrusts AI. Sanjay (ai_trust=0.04) has 72 decisions with 19 positive and 2 negative experiences. These agents are likely using AI outputs informally (checking AI recommendations before making their own decisions) without formally accepting them, which explains the disconnect between positive experiences and low trust.",
        "hypothesis": "These agents are using AI as a 'second opinion' but not formally accepting its recommendations. This allows them to maintain professional autonomy while still benefiting from AI insights. However, this informal use is not captured in the formal AI usage metrics, creating a gap between actual and reported AI adoption.",
        "confidence": 0.85,
        "improvement": 0.0,
        "consequence": 0.7,
        "evidence_sprints": [
          7,
          10,
          11
        ],
        "sprints_in_phase": 6
      },
      "A-S07-P07": {
        "phase": "impact",
        "type": "friction",
        "description": "The bottleneck has fully migrated from AI Pipeline to human agents. AI Pipeline processes simple claims instantly (decisions 0-3, 10, 13, 21-23), but human agents like Diana (260 decisions), Tommy (283 decisions), and Alicia (331 decisions) are overloaded. Diana's exhaustion is 8.0, and she's making decisions on complex claims that require translation (translation=True on multiple decisions).",
        "hypothesis": "The AI Pipeline has become too efficient at processing simple claims, leaving only complex claims for human agents. However, the human workflow hasn't been redesigned to handle this new mix. Agents are now spending most of their time on complex claims that require deep analysis, but they're still using processes designed for a mix of simple and complex claims.",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.85,
        "evidence_sprints": [
          7
        ],
        "sprints_in_phase": 7
      },
      "A-S08-P00": {
        "phase": "impact",
        "type": "friction",
        "description": "Cycle time is reported as 0.0, which is mathematically impossible for a system that is processing claims. This is the third consecutive sprint (Sprint 7, 8) where this metric has been zero, indicating a systemic data pipeline failure rather than a transient glitch. The system is actively processing claims (evidenced by 378 decisions from Alicia alone), so the metric is not reflecting reality.",
        "hypothesis": "The telemetry system that calculates cycle time is likely disconnected from the actual workflow engine, or the calculation logic is broken (e.g., dividing by zero or using an incorrect timestamp field). This is a data integrity issue, not a process issue.",
        "confidence": 0.95,
        "improvement": 0.0,
        "consequence": 0.85,
        "evidence_sprints": [
          8
        ],
        "sprints_in_phase": 6
      },
      "A-S08-P01": {
        "phase": "impact",
        "type": "friction",
        "description": "Handoff failure rate has spiked to 64.0%, a dramatic increase from previous sprints. This is occurring simultaneously with a translation debt index of 14.94 and a first-pass accuracy of only 36%. The combination suggests that the system is failing at the interfaces between AI and human processing, with information being lost or corrupted during transitions.",
        "hypothesis": "The high handoff failure rate is likely caused by AI systems making decisions that human agents cannot validate or understand. When AI auto-processes claims (as seen in the AI Pipeline decisions), the human agents receiving those claims lack the context or confidence to proceed, leading to failed handoffs. The low first-pass accuracy (36%) suggests that the AI is either making incorrect decisions or the human reviewers are rejecting valid AI decisions due to lack of trust.",
        "confidence": 0.88,
        "improvement": 0.0,
        "consequence": 0.95,
        "evidence_sprints": [
          8
        ],
        "sprints_in_phase": 6
      },
      "A-S08-P02": {
        "phase": "theoretical",
        "type": "friction",
        "description": "There is a stark bifurcation in AI trust among high-volume decision-makers. Agents with high decision counts show extreme trust values: Jordan (200 decisions, trust=1.00), Alicia (378 decisions, trust=0.87), Greg (179 decisions, trust=0.88), and Nick (256 decisions, trust=0.00). Meanwhile, low-volume agents (Ron, Tricia, Mike, Leslie with 0 decisions) show moderate trust levels. This suggests that experience with AI is polarizing trust rather than building consensus.",
        "hypothesis": "Agents who have had positive AI experiences (Jordan, Alicia, Greg) are becoming increasingly reliant on AI, while those with negative experiences (Nick, Sanjay, Pat) are becoming completely distrustful. The lack of middle-ground trust suggests that AI errors are catastrophic when they occur, rather than being minor and correctable. This is creating an 'all-or-nothing' trust dynamic.",
        "confidence": 0.82,
        "improvement": 0.0,
        "consequence": 0.78,
        "evidence_sprints": [
          8
        ],
        "sprints_in_phase": 9
      },
      "A-S08-P03": {
        "phase": "replicable",
        "type": "friction",
        "description": "There is a clear correlation between high decision volume and high exhaustion levels. Diana (296 decisions, exhaustion=8.0), Nick (256 decisions, exhaustion=9.0), and Tommy (314 decisions, exhaustion=4.0) show that high-volume agents are experiencing significant fatigue. Notably, Diana has exhaustion=8.0 with only 296 decisions, while Alicia has 378 decisions with exhaustion=3.0, suggesting that the type of decisions matters more than the count.",
        "hypothesis": "Diana and Nick are likely handling more complex or emotionally draining claims (liability determinations, subrogation) compared to Alicia who may be handling simpler FNOL tasks. The exhaustion metric may be reflecting cognitive load rather than raw volume. Diana's low AI trust (0.00) combined with high exhaustion suggests she is manually reviewing everything, while Alicia's high AI trust (0.87) allows her to delegate more to AI.",
        "confidence": 0.75,
        "improvement": 0.0,
        "consequence": 0.82,
        "evidence_sprints": [
          8
        ],
        "sprints_in_phase": 7
      },
      "A-S08-P04": {
        "phase": "theoretical",
        "type": "friction",
        "description": "The shadow_ai pattern from previous sprints persists. Pat (ai_trust=0.20) has 91 decisions with +17/-5 AI experience, and Sanjay (ai_trust=0.05) has 80 decisions with +26/-3 experience. Despite having positive AI experiences (more correct than incorrect), both agents maintain very low AI trust, suggesting they are either not using AI recommendations or are actively working around them.",
        "hypothesis": "These agents may have had a few early negative experiences that created a lasting distrust, or they may be in roles where AI recommendations are less useful. The positive experience ratio (Pat: 17/5, Sanjay: 26/3) suggests that when they do use AI, it works well, but they are choosing not to use it. This could be due to organizational culture, personal preference, or a belief that their manual review is superior.",
        "confidence": 0.85,
        "improvement": 0.0,
        "consequence": 0.72,
        "evidence_sprints": [
          8
        ],
        "sprints_in_phase": 9
      },
      "A-S08-P05": {
        "phase": "impact",
        "type": "friction",
        "description": "The bottleneck has fully migrated to human agents. AI Pipeline processes simple claims automatically (as seen in decisions 1-3, 9-10, 21-23), but human agents like Tommy, Diana, and Pat are handling subrogation, liability determination, and coverage verification manually. The high handoff failure rate (64%) suggests that the AI-to-human handoff is the primary bottleneck.",
        "hypothesis": "The AI system has been optimized for simple claims, leaving complex claims to human agents. However, the handoff between AI and humans is failing because the AI does not provide sufficient context or reasoning for its decisions, leaving human agents to re-verify everything. This creates a bottleneck where human agents are doing redundant work.",
        "confidence": 0.8,
        "improvement": 0.0,
        "consequence": 0.88,
        "evidence_sprints": [
          8
        ],
        "sprints_in_phase": 6
      },
      "A-S08-P06": {
        "phase": "replicable",
        "type": "friction",
        "description": "Exception rate has risen to 7.14% from 6.37% in Sprint 7, continuing an upward trend. The decision log shows Tommy escalating a subrogation case (decision 15) by requesting more information, and Diana handling liability determinations that may require escalation. The combination of rising exceptions and high handoff failures suggests the system is becoming less stable.",
        "hypothesis": "The rising exception rate may be caused by AI systems incorrectly classifying claims as simple when they are actually complex, leading to human agents discovering issues that should have been caught earlier. The high handoff failure rate compounds this by creating more exceptions as agents struggle to process incomplete or incorrect AI handoffs.",
        "confidence": 0.7,
        "improvement": 0.0,
        "consequence": 0.8,
        "evidence_sprints": [
          8
        ],
        "sprints_in_phase": 7
      },
      "A-S08-P07": {
        "phase": "impact",
        "type": "friction",
        "description": "Translation debt index is at 14.94, indicating significant meaning loss at handoffs. This is directly correlated with the 64% handoff failure rate. The decision log shows AI Pipeline decisions that are brief ('AI auto-processed...') without providing the reasoning or context that human agents need to understand the decision.",
        "hypothesis": "The AI system is not providing sufficient context in its handoffs. When AI auto-processes a claim, it only records the outcome, not the reasoning. Human agents receiving these claims must reconstruct the reasoning from scratch, leading to information loss and failed handoffs.",
        "confidence": 0.85,
        "improvement": 0.0,
        "consequence": 0.9,
        "evidence_sprints": [
          8
        ],
        "sprints_in_phase": 6
      },
      "A-S09-P00": {
        "phase": "impact",
        "type": "friction",
        "description": "Translation debt index has dropped to 10.26 from 14.94 in Sprint 8, but the pattern persists in human decision-making. Diana's decisions show a high concentration of translation=True flags (decisions 9, 10, 14, 16, 22), indicating she is consistently translating complex claim context into simplified approval/escalate actions. The handoff failure rate has spiked to 60%, suggesting that despite the lower translation index, the actual meaning loss at handoffs is worsening.",
        "hypothesis": "The translation debt index dropped because AI is auto-processing more simple claims, but the remaining human-handled claims are increasingly complex. Diana, as the highest-volume human agent, is absorbing the most complex cases and her translations are losing critical nuance (e.g., 'genuinely disputed liability' vs. 'unresolved liability'). The 60% handoff failure rate indicates that downstream agents cannot reconstruct the full claim context from her simplified outputs.",
        "confidence": 0.85,
        "improvement": 0.0,
        "consequence": 0.95,
        "evidence_sprints": [
          9
        ],
        "sprints_in_phase": 5
      },
      "A-S09-P01": {
        "phase": "impact",
        "type": "friction",
        "description": "A dangerous bifurcation is emerging: high-volume agents (Diana, Tommy, Nick, Sanjay) have very low AI trust (0.05, 0.00, 0.00, 0.00 respectively), while low-volume agents (Kathryn, Tricia, Jordan, Greg, Alicia) have high trust (0.82, 0.79, 1.00, 0.90, 0.84). The high-volume agents are making 340, 349, 292, and 94 decisions respectively, while high-trust agents make 21, 0, 225, 200, and 425 decisions. This suggests that agents who interact most with the AI are losing trust, while those who use it less maintain positive views.",
        "hypothesis": "High-volume agents are encountering edge cases and complex claims that the AI handles poorly. Diana's +0/-15 AI experience shows she has seen 15 AI failures with zero successes, likely because she is assigned the most complex claims where AI recommendations are unreliable. Low-volume agents like Jordan and Alicia are processing simpler claims where AI performs well, reinforcing their trust. This creates a self-reinforcing loop where the most experienced agents reject AI, making the system less effective overall.",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.9,
        "evidence_sprints": [
          9
        ],
        "sprints_in_phase": 5
      },
      "A-S09-P02": {
        "phase": "impact",
        "type": "friction",
        "description": "Exception rate has risen to 8.33% from 7.14% in Sprint 8, continuing the upward trend. The exception pattern is concentrated in human decisions, particularly Diana (decision 9: complex 4-vehicle injury claim) and Tommy (decision 6: complex rear-end collision). The supplement_request_rate of 24% suggests that agents are frequently requesting additional information, which may be a form of exception avoidance or a symptom of incomplete information at handoffs.",
        "hypothesis": "The combination of high handoff failure rate (60%) and translation debt is causing agents to receive incomplete claim information, leading them to escalate or request supplements rather than make decisions. The exception rate is a downstream symptom of poor information flow, not a reflection of genuinely exceptional claims.",
        "confidence": 0.8,
        "improvement": 0.0,
        "consequence": 0.85,
        "evidence_sprints": [
          9
        ],
        "sprints_in_phase": 5
      },
      "A-S09-P03": {
        "phase": "theoretical",
        "type": "friction",
        "description": "The shadow_ai pattern persists and is now more pronounced. Pat (ai_trust=0.07) has 106 decisions with +11/-1 AI experience, yet still uses AI in decision 20 (coverage_verification_human \u2192 approve with AI used=True). Sanjay (ai_trust=0.00) has 94 decisions with +21/-0 AI experience, showing he uses AI but doesn't trust it. This suggests agents are using AI outputs as a reference but making their own decisions, creating a parallel decision-making process.",
        "hypothesis": "Agents are using AI as a second opinion but not integrating it into their decision framework. The high AI success rate (Pat: +11, Sanjay: +21) should build trust, but the agents' low trust scores suggest they are not internalizing AI successes. This may be because they are not receiving feedback on AI correctness, or because they attribute successes to their own judgment rather than AI assistance.",
        "confidence": 0.85,
        "improvement": 0.0,
        "consequence": 0.75,
        "evidence_sprints": [
          9
        ],
        "sprints_in_phase": 8
      },
      "A-S09-P04": {
        "phase": "impact",
        "type": "friction",
        "description": "The bottleneck has fully migrated to human agents, but with a new twist: Diana is now the primary bottleneck with 340 decisions and exhaustion=8.0, while AI Pipeline handles simple claims automatically. However, the cycle_time of 0.0 suggests that claims are not being delayed in the system, which is contradictory. This may indicate that the bottleneck is in decision quality rather than throughput.",
        "hypothesis": "The cycle_time of 0.0 is misleading because it only measures throughput, not quality. Diana is processing claims quickly but with poor accuracy (40% first_pass), meaning many claims will need rework downstream. The bottleneck has shifted from 'waiting for processing' to 'processing incorrectly', which is more expensive because it requires rework.",
        "confidence": 0.8,
        "improvement": 0.0,
        "consequence": 0.9,
        "evidence_sprints": [
          9
        ],
        "sprints_in_phase": 5
      },
      "A-S09-P05": {
        "phase": "impact",
        "type": "friction",
        "description": "The correlation between decision volume and exhaustion persists and is now more severe. Diana (340 decisions, exhaustion=8.0) and Nick (292 decisions, exhaustion=9.0) show critical exhaustion levels. However, Alicia (425 decisions, exhaustion=3.0) and Jordan (225 decisions, exhaustion=2.0) show that high volume doesn't necessarily lead to exhaustion when AI trust is high. This suggests exhaustion is driven by cognitive load from distrust, not just volume.",
        "hypothesis": "Agents with low AI trust must manually verify every AI recommendation, doubling their cognitive load. Diana and Nick are not just processing claims; they are also fighting the AI system, which is exhausting. Alicia and Jordan trust AI and can process claims more efficiently, reducing cognitive load despite higher volume.",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.85,
        "evidence_sprints": [
          9
        ],
        "sprints_in_phase": 5
      },
      "A-S09-P06": {
        "phase": "theoretical",
        "type": "friction",
        "description": "A pidgin language is emerging in agent decision rationales. Diana's decisions 14 and 16 use nearly identical language ('Liability is clear, no injuries, single vehicle') despite being different claims. This suggests agents are developing shorthand that loses claim-specific detail. The translation=True flags on these decisions confirm that meaning is being compressed.",
        "hypothesis": "Agents are under time pressure and have developed standardized phrases to quickly document their decisions. While this speeds up documentation, it strips away claim-specific nuances that downstream agents need. The pidgin is a coping mechanism for high workload, but it degrades information quality.",
        "confidence": 0.75,
        "improvement": 0.0,
        "consequence": 0.7,
        "evidence_sprints": [
          9
        ],
        "sprints_in_phase": 8
      },
      "A-S10-P01": {
        "phase": "impact",
        "type": "friction",
        "description": "Diana remains the primary bottleneck with 370 decisions and exhaustion at 8.0, but now shows near-zero stress (0.05) and near-zero AI trust (0.08). She is processing 5x more decisions than the next human (Tommy at 392 is close, but Diana's are all complex claims). Her exhaustion is critical while her stress is artificially low, suggesting she has disengaged from quality concerns.",
        "hypothesis": "Diana has become the de facto expert for complex claims, but her low AI trust means she rejects all AI assistance. Her low stress despite high exhaustion suggests she has stopped caring about the cognitive load and is just pushing through, possibly cutting corners on quality to maintain throughput.",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.95,
        "evidence_sprints": [
          10
        ],
        "sprints_in_phase": 4
      },
      "A-S10-P02": {
        "phase": "impact",
        "type": "friction",
        "description": "A severe trust bifurcation is emerging: high-volume agents (Alicia, Jordan, Greg, Nick) have AI trust 0.0-1.0 with high positive experience ratios, while low-volume or bottleneck agents (Diana, Pat, Sanjay, Tommy) have low trust with mixed experiences. Sanjay has 101 decisions, 32 positive and 2 negative AI experiences, yet trust is 0.00 \u2014 indicating active distrust despite positive evidence.",
        "hypothesis": "Trust is not being updated based on experience but is instead driven by organizational role and perceived threat. Sanjay and Diana may see AI as a threat to their expertise or job security, so they discount positive experiences. High-trust agents may be in roles where AI reduces their workload without threatening their identity.",
        "confidence": 0.8,
        "improvement": 0.0,
        "consequence": 0.85,
        "evidence_sprints": [
          10
        ],
        "sprints_in_phase": 4
      },
      "A-S10-P03": {
        "phase": "replicable",
        "type": "friction",
        "description": "Exception rate has risen to 3.92% (from 8.33% last sprint \u2014 actually decreased, but the sample shows a critical exception: the settlement_ai translation failure). Handoff failure rate is 16%, which is extremely high and likely driving the exception rate. The exception in decision 5 is a pipeline failure, not a human judgment call.",
        "hypothesis": "The 16% handoff failure rate suggests that 1 in 6 AI-to-human or human-to-human handoffs is failing, likely due to incomplete information transfer. The translation debt is the root cause, and the exception rate is the symptom. The decrease in exception rate from last sprint may be because agents are now pre-emptively escalating rather than letting the pipeline fail.",
        "confidence": 0.75,
        "improvement": 0.0,
        "consequence": 0.8,
        "evidence_sprints": [
          10
        ],
        "sprints_in_phase": 5
      },
      "A-S10-P05": {
        "phase": "impact",
        "type": "friction",
        "description": "The correlation between decision volume and exhaustion is now extreme: Diana (370 decisions, exhaustion 8.0), Nick (321 decisions, exhaustion 9.0), Tommy (392 decisions, exhaustion 4.0 \u2014 anomaly). Nick has high volume and high exhaustion but zero AI trust and zero AI usage, suggesting he is doing everything manually and burning out.",
        "hypothesis": "Nick is processing high volume but refusing AI assistance (trust=0.00), leading to extreme exhaustion. His stress is 0.00, which is concerning \u2014 he may be dissociating or has given up on self-care. Tommy's low exhaustion despite high volume suggests he is effectively using AI (23 positive experiences) but not acknowledging it in trust.",
        "confidence": 0.8,
        "improvement": 0.0,
        "consequence": 0.9,
        "evidence_sprints": [
          10
        ],
        "sprints_in_phase": 4
      },
      "A-S10-P06": {
        "phase": "theoretical",
        "type": "friction",
        "description": "The pidgin language pattern persists, but now shows a new dimension: agents are using AI-generated phrases in their human decision rationales. Tommy's decision 7 says 'This is a straightforward auto claim with clear liability, no injuries, and PA j...' which is nearly identical to AI-generated text in decision 16. This is not just shared vocabulary \u2014 it's shared sentence structure.",
        "hypothesis": "Agents are increasingly copying AI-generated rationales into their own decision notes, either because they agree with the AI and see no reason to rewrite, or because they are using AI output as a template to speed up their documentation. This reduces the informational value of human decisions and makes it harder to distinguish human judgment from AI output.",
        "confidence": 0.65,
        "improvement": 0.0,
        "consequence": 0.7,
        "evidence_sprints": [
          10
        ],
        "sprints_in_phase": 7
      },
      "A-S10-P07": {
        "phase": "impact",
        "type": "friction",
        "description": "Cycle time is reported as 0.0, which is impossible for a claims process with human involvement. This suggests the metric is being gamed or misreported. Cost per claim is $234.04, which is high, and first-pass accuracy is 76%, which is low. The zero cycle time may be masking severe delays that are being absorbed by agents working overtime.",
        "hypothesis": "Cycle time is being measured from AI pipeline start to AI pipeline end, excluding human processing time. This makes the metric look perfect while hiding the real bottleneck. The high cost and low accuracy suggest that human rework is significant, but it's not being captured in the cycle time metric.",
        "confidence": 0.8,
        "improvement": 0.0,
        "consequence": 0.85,
        "evidence_sprints": [
          10
        ],
        "sprints_in_phase": 4
      },
      "A-S10-P09": {
        "phase": "replicable",
        "type": "friction",
        "description": "A new pattern emerges: agents with the highest AI trust (Jordan, Alicia, Greg at 1.00) have zero negative AI experiences, while agents with moderate trust (Pat, Tommy) have negative experiences but continue using AI. This creates a paradox where the most trusting agents have never seen AI fail, making their trust fragile and potentially dangerous.",
        "hypothesis": "High-trust agents have been assigned only simple claims where AI is highly reliable, so they have never encountered AI failure. Low-trust agents have been assigned more complex claims where AI fails more often, creating a self-reinforcing cycle: high-trust agents get easy work, low-trust agents get hard work, and neither group's trust reflects the true AI capability distribution.",
        "confidence": 0.75,
        "improvement": 0.0,
        "consequence": 0.9,
        "evidence_sprints": [
          10
        ],
        "sprints_in_phase": 5
      },
      "A-S11-P00": {
        "phase": "impact",
        "type": "friction",
        "description": "Diana remains the primary human bottleneck, but the pattern has shifted from decision volume to exhaustion-driven quality risk. Diana has 412 decisions (highest in the org), exhaustion=8.0, stress=0.00, and ai_trust=0.00. She is processing complex claims manually (decisions 1,2,3,4,6,7,8,9,10,13,19) while AI handles only simple claims. Her exhaustion is at the critical threshold while her stress reads 0.00, suggesting she has disengaged from the AI system entirely and is operating on manual override.",
        "hypothesis": "Diana has been assigned the most complex claims while AI auto-processes simple ones. Her zero AI trust and zero stress suggest she has mentally checked out of the AI collaboration model and is operating as a fully manual processor. The exhaustion=8.0 with stress=0.00 indicates she may be experiencing burnout-induced apathy rather than active resistance.",
        "confidence": 0.92,
        "improvement": 0.0,
        "consequence": 0.85,
        "evidence_sprints": [
          11
        ],
        "sprints_in_phase": 3
      },
      "A-S11-P02": {
        "phase": "impact",
        "type": "friction",
        "description": "A bifurcated trust landscape has emerged. Three agents (Jordan, Greg, Alicia) show high trust (0.76-1.00) with positive AI experiences. Four agents (Diana, Sanjay, Nick, Rachel) show zero or near-zero trust (0.00-0.18). The zero-trust agents have high decision volumes (Diana=412, Nick=349, Sanjay=112, Rachel=153) and high exhaustion (Diana=8, Nick=9, Sanjay=7). This is not a uniform adoption problem but a targeted trust collapse among high-volume processors.",
        "hypothesis": "High-volume agents are experiencing AI as a workload amplifier rather than a relief mechanism. They process too many decisions to meaningfully evaluate AI recommendations, leading to either blind acceptance (Nick's +24/-0) or complete rejection (Diana's 0 AI usage). The zero-trust agents have high exhaustion, suggesting they view AI as adding cognitive load rather than reducing it.",
        "confidence": 0.88,
        "improvement": 0.0,
        "consequence": 0.9,
        "evidence_sprints": [
          11
        ],
        "sprints_in_phase": 3
      },
      "A-S11-P03": {
        "phase": "theoretical",
        "type": "friction",
        "description": "Exception rate is 4.43% while supplement request rate is 16.0%. The exception rate is moderate but the supplement request rate is nearly 4x higher, suggesting that agents are not formally escalating exceptions but are informally requesting more information. This indicates a hidden exception layer that isn't being captured in formal metrics.",
        "hypothesis": "Agents are avoiding formal exception escalation (which would increase exception_rate) and instead using supplement requests as an informal workaround. This keeps formal metrics low but creates hidden coordination_drag. The 16% supplement rate suggests agents are spending significant time on information gathering rather than decision-making.",
        "confidence": 0.71,
        "improvement": 0.0,
        "consequence": 0.68,
        "evidence_sprints": [
          11
        ],
        "sprints_in_phase": 6
      },
      "A-S11-P04": {
        "phase": "impact",
        "type": "friction",
        "description": "Cycle time is reported as 0.0, which is impossible given 15 human agents making 412+ decisions with handoff failures and supplement requests. This metric has been 0.0 for multiple sprints (persistence: 6), suggesting the metric is either not being measured correctly or is being gamed. The cost_per_claim of $268.92 with first_pass_accuracy of 76% suggests real work is happening, but cycle time is not capturing it.",
        "hypothesis": "The cycle time metric is either not instrumented correctly or is being reported as 0.0 to mask operational issues. The combination of 8% handoff failures and 16% supplement requests would necessarily create non-zero cycle time. This metric inversion hides the true cost of coordination_drag.",
        "confidence": 0.95,
        "improvement": 0.0,
        "consequence": 0.82,
        "evidence_sprints": [
          11
        ],
        "sprints_in_phase": 3
      },
      "A-S11-P05": {
        "phase": "theoretical",
        "type": "friction",
        "description": "Rachel has 153 decisions, 0 positive AI experiences, 0 negative, and ai_trust=0.18. She is making decisions without any AI interaction, suggesting she is either bypassing the AI system entirely or the system is not presenting recommendations to her. Her low trust (0.18) with zero experience indicates she may be operating on preconceived notions rather than evidence.",
        "hypothesis": "Rachel may be receiving informal guidance from Diana or other zero-trust agents. Her non-zero trust with zero experience suggests she has formed opinions based on others' experiences rather than her own. This indicates informal_control_exposure where negative sentiment spreads through social channels.",
        "confidence": 0.64,
        "improvement": 0.0,
        "consequence": 0.55,
        "evidence_sprints": [
          11
        ],
        "sprints_in_phase": 6
      },
      "A-S11-P07": {
        "phase": "theoretical",
        "type": "friction",
        "description": "The pidgin language pattern persists (persistence: 5) and now shows a new dimension: agents are using AI-generated phrasing in their manual decisions. Tommy's decision 24 states 'The claim is a simple, clear-liability auto claim in PA with no injuries. The AI' - this appears to be AI-generated text that Tommy is adopting. This suggests agents are internalizing AI language patterns even when making manual decisions.",
        "hypothesis": "Agents are learning AI language patterns through repeated exposure and adopting them in their own decision justifications. This is a form of pidgin emergence where the boundary between human and AI language blurs. While this may improve consistency, it also risks losing human nuance in complex cases.",
        "confidence": 0.81,
        "improvement": 0.0,
        "consequence": 0.63,
        "evidence_sprints": [
          11
        ],
        "sprints_in_phase": 6
      },
      "A-S11-P08": {
        "phase": "theoretical",
        "type": "friction",
        "description": "The combination of 8% handoff failure rate, 16% supplement request rate, and 4.43% exception rate indicates significant coordination overhead. With 15 agents and multiple handoffs per claim (FNOL \u2192 investigation \u2192 coverage \u2192 liability \u2192 settlement \u2192 approval \u2192 subrogation), the probability of at least one failure per claim is high. The cost_per_claim of $268.92 may be inflated by this coordination drag.",
        "hypothesis": "Each handoff between agents (or between AI and human) introduces a failure probability. With 6-7 handoffs per claim, the cumulative failure rate is 1-(0.92^6) = 39%. This means nearly 4 in 10 claims experience at least one coordination failure, driving up costs and cycle time (though masked at 0.0).",
        "confidence": 0.74,
        "improvement": 0.0,
        "consequence": 0.7,
        "evidence_sprints": [
          11
        ],
        "sprints_in_phase": 6
      },
      "A-S12-P00": {
        "phase": "impact",
        "type": "friction",
        "description": "The system is operating at 'bounded_initiative' capability, yet the human agents are making 449, 494, 565, and 383 decisions respectively (Diana, Tommy, Alicia, Nick) while the AI pipeline is only auto-processing simple claims. The human agents are carrying the full cognitive load for complex claims, but the system's capability level suggests the AI should be taking more initiative. The exception rate of 2.67% and supplement request rate of 4.0% indicate the human agents are catching issues the AI cannot handle, but the AI is not being given the authority to handle more complex cases.",
        "hypothesis": "The 'bounded_initiative' capability level is a constraint that prevents the AI from taking on more complex claims, even though the human agents are demonstrating they can handle the volume. The system is not expanding the AI's authority to match the demonstrated capability of the human-AI team.",
        "confidence": 0.85,
        "improvement": 0.0,
        "consequence": 0.9,
        "evidence_sprints": [
          12
        ],
        "sprints_in_phase": 2
      },
      "A-S12-P01": {
        "phase": "impact",
        "type": "friction",
        "description": "There is a stark correlation between high exhaustion and zero AI trust. Diana (exhaustion=8.0, ai_trust=0.00), Nick (exhaustion=9.0, ai_trust=0.00), and Sanjay (exhaustion=7.0, ai_trust=0.00) all show complete distrust of AI despite having high decision volumes. Meanwhile, agents with low exhaustion (Jordan=2.0, Tricia=2.0, Alicia=3.0) have high AI trust (1.00, 0.79, 0.69). This suggests exhaustion is driving distrust, not the other way around.",
        "hypothesis": "Exhausted agents are more likely to see AI as an additional burden rather than a helper. They may have had negative experiences with AI recommendations that added to their workload, or they may simply lack the cognitive bandwidth to evaluate AI suggestions. The high decision volume combined with exhaustion creates a 'trust deficit spiral' where exhausted agents reject AI, do more work manually, get more exhausted, and trust AI even less.",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.85,
        "evidence_sprints": [
          12
        ],
        "sprints_in_phase": 2
      },
      "A-S12-P02": {
        "phase": "impact",
        "type": "friction",
        "description": "Pat and Sanjay show a pattern of high decision volume with zero AI trust despite having positive AI experiences. Pat has 143 decisions, 13 positive AI experiences, 0 negative, but ai_trust=0.00. Sanjay has 124 decisions, 26 positive AI experiences, 0 negative, but ai_trust=0.00. This is a more extreme version of the shadow_ai pattern seen in Sprint 11, where agents are using AI outputs but not trusting them enough to acknowledge the AI's contribution.",
        "hypothesis": "These agents are likely using AI outputs as a 'second opinion' but not integrating them into their decision-making process. They may be re-doing the AI's work to verify it, which doubles their workload. The positive AI experiences are not translating to trust because the agents don't perceive the AI as saving them time\u2014they see it as adding an extra step.",
        "confidence": 0.95,
        "improvement": 0.0,
        "consequence": 0.8,
        "evidence_sprints": [
          12
        ],
        "sprints_in_phase": 2
      },
      "A-S12-P03": {
        "phase": "impact",
        "type": "friction",
        "description": "Sanjay's approval decision (decision #15) shows a critical authority ambiguity: 'I cannot approve this $81K claim. The file is incomplete. Despite multiple requests...' This is a high-value claim that is stuck because the approval authority is unclear. The AI Pipeline is auto-processing approvals for simple claims ($2,553.71), but the complex $81K claim is stuck in human review with no clear escalation path.",
        "hypothesis": "The approval process has a gap between what the AI can auto-approve (simple claims) and what humans can approve (complex claims). For complex claims, the authority to approve is ambiguous\u2014Sanjay is requesting information but may not have the authority to approve the claim even if the information is provided. This creates a bottleneck for high-value claims.",
        "confidence": 0.8,
        "improvement": 0.0,
        "consequence": 0.9,
        "evidence_sprints": [
          12
        ],
        "sprints_in_phase": 2
      },
      "A-S12-P04": {
        "phase": "theoretical",
        "type": "friction",
        "description": "The pidgin language pattern persists, but now shows a new dimension: agents are using AI acceptance as a form of communication. Tommy's decision #24 shows 'This simple auto claim has clear liability, no injuries, and a single vehicle, w'\u2014he is using the AI's language patterns to justify his acceptance. Jordan's decision #20 shows 'The AI recommendation aligns with the standard routing for this simple auto clai'\u2014again using AI-style language. This suggests agents are learning to speak 'AI' to justify their decisions, which may mask genuine disagreement.",
        "hypothesis": "Agents are learning that using AI-style language ('simple claim', 'clear liability', 'single vehicle') makes their decisions more acceptable to the system. This is a form of gaming the system\u2014they may be accepting AI recommendations not because they agree, but because it's easier than explaining a disagreement. This could mask genuine concerns about AI accuracy.",
        "confidence": 0.75,
        "improvement": 0.0,
        "consequence": 0.7,
        "evidence_sprints": [
          12
        ],
        "sprints_in_phase": 5
      },
      "A-S12-P05": {
        "phase": "theoretical",
        "type": "friction",
        "description": "A new pattern emerges: agents with high AI trust and high positive AI experience are NOT the ones with the highest decision volumes. Jordan (ai_trust=1.00, +13/-0, decisions=300) and Greg (ai_trust=0.93, +13/-2, decisions=266) have moderate decision volumes, while Alicia (ai_trust=0.69, +24/-7, decisions=565) has the highest decision volume but lower trust. This suggests that high decision volume is not building trust\u2014it's eroding it, even with positive AI experiences.",
        "hypothesis": "The relationship between AI experience and trust is not linear. At high decision volumes, the negative experiences (Alicia has 7 negative) have a disproportionate impact on trust. Additionally, high-volume agents may be seeing AI errors that lower-volume agents don't encounter, because they're processing more edge cases.",
        "confidence": 0.8,
        "improvement": 0.0,
        "consequence": 0.75,
        "evidence_sprints": [
          12
        ],
        "sprints_in_phase": 5
      },
      "A-S13-P01": {
        "phase": "impact",
        "type": "friction",
        "description": "Diana shows a critical pattern: exhaustion=8.0, ai_trust=0.00, but with 478 decisions and +5/-0 AI experience. She is making the most decisions of any human agent, has zero trust in AI despite positive experiences, and is severely exhausted. This is a high-risk burnout and quality failure point. Nick (exhaustion=9.0, ai_trust=0.00, 423 decisions) shows a similar pattern, though with +12/-0 AI experience.",
        "hypothesis": "Diana and Nick are being overloaded with high-volume, high-complexity claims that require human judgment. Despite AI being correct in their experiences, they maintain zero trust, likely because they are seeing edge cases or failure modes not captured in the +5/-0 metric. Their exhaustion suggests they are working beyond sustainable capacity, which erodes trust in any automated system that might add to their cognitive load, even if it's correct.",
        "confidence": 0.88,
        "improvement": 0.0,
        "consequence": 0.9,
        "evidence_sprints": [
          13
        ],
        "sprints_in_phase": 1
      },
      "A-S13-P02": {
        "phase": "impact",
        "type": "friction",
        "description": "Pat (153 decisions, ai_trust=0.05, +18/-3) and Sanjay (131 decisions, ai_trust=0.01, +30/-4) continue to show high decision volume with near-zero AI trust despite having the most positive AI experiences of any agents. They are actively working around the AI system, making their own decisions rather than accepting AI recommendations. Sanjay's decision #7 shows him requesting information on a complex claim that AI likely could have assisted with.",
        "hypothesis": "Pat and Sanjay are likely seeing AI failures that are not captured in the +18/-3 and +30/-4 metrics. The -3 and -4 negative experiences may be disproportionately impactful, especially if they occurred early in their AI adoption. They may also be handling claims that are systematically different from what AI is trained on, making AI recommendations unreliable for their specific portfolio. Their high decision volume suggests they are being assigned the most complex claims, which AI may not handle well.",
        "confidence": 0.85,
        "improvement": 0.0,
        "consequence": 0.82,
        "evidence_sprints": [
          13
        ],
        "sprints_in_phase": 1
      },
      "A-S13-P03": {
        "phase": "theoretical",
        "type": "friction",
        "description": "A clear positive trust cascade is emerging among agents with lower decision volumes. Jordan (ai_trust=1.00, 325 decisions, +18/-0), Alicia (ai_trust=0.98, 609 decisions, +21/-0), Greg (ai_trust=1.00, 286 decisions, +11/-0), and Kathryn (ai_trust=0.75, 24 decisions) all show high trust with zero negative AI experiences. These agents are accepting AI recommendations at high rates and are not experiencing translation failures or exceptions.",
        "hypothesis": "These agents are likely handling simpler, more standardized claims where AI performs well. Their positive experiences reinforce trust, creating a virtuous cycle. They are also likely receiving AI recommendations that are well-aligned with their own judgment, reducing cognitive dissonance. Their lower exhaustion levels (2-5) suggest they are not overwhelmed, allowing them to evaluate AI recommendations more objectively.",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.7,
        "evidence_sprints": [
          13
        ],
        "sprints_in_phase": 4
      },
      "A-S13-P04": {
        "phase": "theoretical",
        "type": "friction",
        "description": "The approval stage shows signs of authority ambiguity. Sanjay's decision #7 shows him requesting information on a complex claim, while Diana's decisions #8 and #14 show her manually approving claims after reviewing AI assessments. The AI pipeline is auto-processing approvals for simple claims (decisions #2, #12, #13, #15, #18, #23), but there's no clear pattern for who has authority to override AI in complex cases. This creates a bottleneck where complex claims may stall waiting for human approval.",
        "hypothesis": "There is no clear escalation path for complex claims that AI cannot handle. The AI pipeline auto-approves simple claims, but for complex claims, the system relies on individual agents to decide whether to accept AI recommendations or override them. This ambiguity leads to inconsistent handling: some agents (like Diana) manually review everything, while others (like Sanjay) request more information, potentially causing delays.",
        "confidence": 0.75,
        "improvement": 0.0,
        "consequence": 0.78,
        "evidence_sprints": [
          13
        ],
        "sprints_in_phase": 4
      },
      "A-S13-P05": {
        "phase": "impact",
        "type": "friction",
        "description": "First pass accuracy is 68%, but the handoff_failure_rate is 24% and translation_debt_index is 4.67. This suggests that the 'first pass' metric is misleading \u2014 while 68% of claims are processed correctly on the first pass, a significant portion of the remaining 32% are failing due to translation issues at handoffs, not due to incorrect AI decisions. The AI is making correct decisions (as evidenced by high positive AI experiences), but the handoff layer is corrupting the process.",
        "hypothesis": "The first_pass_accuracy metric is measuring the AI's decision quality, but the handoff_failure_rate is measuring the system's ability to transmit that decision to the next stage. The high translation debt indicates that the AI's output is not being properly formatted or interpreted at handoffs, causing otherwise correct decisions to be flagged as failures. This is a system integration issue, not an AI quality issue.",
        "confidence": 0.85,
        "improvement": 0.0,
        "consequence": 0.88,
        "evidence_sprints": [
          13
        ],
        "sprints_in_phase": 1
      },
      "A-S13-P06": {
        "phase": "impact",
        "type": "friction",
        "description": "A paradoxical pattern emerges: agents with the most positive AI experiences (Pat +18/-3, Sanjay +30/-4) have the lowest AI trust (0.05 and 0.01), while agents with fewer positive experiences (Jordan +18/-0, Alicia +21/-0) have near-perfect trust. This inverts the expected relationship between experience and trust. The negative experiences (-3 and -4) appear to have a disproportionate impact on trust, outweighing the positive experiences by a factor of 10x or more.",
        "hypothesis": "Negative AI experiences are not weighted equally with positive experiences \u2014 they are weighted disproportionately higher. This is consistent with loss aversion theory in behavioral economics. A single AI failure that causes a significant problem (e.g., a wrong settlement amount) can erase the trust built by dozens of correct decisions. Pat and Sanjay may have experienced early failures that set a negative anchor, and subsequent positive experiences cannot overcome this initial negative impression.",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.85,
        "evidence_sprints": [
          13
        ],
        "sprints_in_phase": 1
      },
      "A-S13-P07": {
        "phase": "theoretical",
        "type": "friction",
        "description": "There is a clear coordination drag pattern among high-volume agents. Diana (478 decisions), Tommy (535 decisions), Alicia (609 decisions), and Nick (423 decisions) are making the majority of decisions, but they have wildly different AI trust levels (0.00, 0.00, 0.98, 0.00). This suggests that the organization is not coordinating AI adoption effectively \u2014 some agents are fully leveraging AI while others are completely ignoring it, even when handling similar claim types.",
        "hypothesis": "The organization lacks a coordinated AI adoption strategy. Agents are left to form their own opinions about AI based on their individual experiences, without a structured framework for evaluating AI recommendations. This leads to a fragmented workforce where AI adoption is determined by individual personality and experience rather than organizational policy. Tommy's +12/-15 experience is particularly concerning \u2014 he has the most negative experiences of any agent, which may be driving his zero trust.",
        "confidence": 0.8,
        "improvement": 0.0,
        "consequence": 0.75,
        "evidence_sprints": [
          13
        ],
        "sprints_in_phase": 4
      },
      "A-S14-P01": {
        "phase": "replicable",
        "type": "friction",
        "description": "A severe trust asymmetry is emerging. High-volume agents (Diana, Tommy, Nick) have ai_trust=0.00, while low-volume agents (Jordan, Greg, Alicia) have ai_trust between 0.90 and 1.00. Diana has made 518 decisions with zero AI usage and zero trust, despite the system being available. Tommy has 576 decisions with 13 positive and 12 negative AI experiences, yet trust is 0.00 \u2014 suggesting negative experiences are disproportionately weighting trust.",
        "hypothesis": "High-volume agents are likely seeing the AI's failures more frequently in aggregate, and the negative experiences (even if proportionally small) are creating a 'one bad apple' effect. Additionally, these agents may have developed their own heuristics and workflows that don't align with AI recommendations, making AI integration feel like an interruption rather than an aid.",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.85,
        "evidence_sprints": [
          14
        ],
        "sprints_in_phase": 1
      },
      "A-S14-P02": {
        "phase": "replicable",
        "type": "friction",
        "description": "Diana remains a severe bottleneck. She has 518 decisions, exhaustion=8.0, and is handling investigation, liability_determination, and settlement for complex claims. Her decisions show she is manually processing complex three-vehicle claims with injuries (decisions #1, #3, #4, #15) while also handling moderate claims. Her exhaustion is at the maximum level (8.0), yet she continues to be assigned the most complex work.",
        "hypothesis": "Diana is the most experienced or senior adjuster, so the system routes the most complex claims to her. However, she refuses to use AI (trust=0.00), meaning she manually processes every step, creating a bottleneck. Her exhaustion is high, which may degrade her decision quality over time.",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.9,
        "evidence_sprints": [
          14
        ],
        "sprints_in_phase": 1
      },
      "A-S14-P03": {
        "phase": "theoretical",
        "type": "friction",
        "description": "A paradoxical pattern emerges where agents with the most negative AI experiences (Tommy: +13/-12) have zero trust, while agents with fewer negative experiences (Alicia: +32/-3, Nick: +18/-1) have either high trust (0.90) or zero trust (0.00). Specifically, Nick has 18 positive and only 1 negative experience but trust=0.00, while Alicia has 32 positive and 3 negative with trust=0.90. This suggests trust is not purely experience-based but influenced by other factors.",
        "hypothesis": "Trust may be influenced by the recency of negative experiences, the severity of the negative outcome, or the agent's pre-existing disposition toward automation. Nick's single negative experience may have been particularly impactful (e.g., a high-value claim error), while Alicia's negative experiences may have been low-stakes. Alternatively, agents with higher decision volumes may have developed stronger heuristics that make them more critical of AI suggestions.",
        "confidence": 0.8,
        "improvement": 0.0,
        "consequence": 0.7,
        "evidence_sprints": [
          14
        ],
        "sprints_in_phase": 3
      },
      "A-S14-P04": {
        "phase": "theoretical",
        "type": "friction",
        "description": "The exception rate is 4.49%, which is lower than previous sprints, but the exceptions that do occur are concentrated in complex claims (three-vehicle, injury, disputed liability). Decision #4 (Diana settlement escalate) and #10 (Tommy subrogation escalate) both involve complex claims. The AI pipeline also generated an exception at settlement (decision #5) due to translation failure. This suggests exceptions are not random but cluster around high-complexity, high-stakes cases.",
        "hypothesis": "Complex claims inherently have more ambiguity and require human judgment, so exceptions are expected. However, the AI pipeline is also generating exceptions on these claims, suggesting the AI is not yet capable of handling multi-vehicle injury cases, and the handoff failure compounds the problem.",
        "confidence": 0.75,
        "improvement": 0.0,
        "consequence": 0.65,
        "evidence_sprints": [
          14
        ],
        "sprints_in_phase": 3
      },
      "A-S14-P05": {
        "phase": "theoretical",
        "type": "friction",
        "description": "First pass accuracy is 76%, which is an improvement from 68% in Sprint 13, but the translation debt index (2.56) and handoff failure rate (16%) suggest that the 'first pass' metric may be misleading. The AI pipeline is auto-processing simple claims correctly (decisions #6, #7, #11, #16), but the handoff failures at settlement (decision #5) are not captured in first_pass_accuracy, meaning the metric overstates true end-to-end accuracy.",
        "hypothesis": "First pass accuracy is measured at the point of decision, but handoff failures occur after the decision, so they are not counted. The metric is therefore not capturing the full cost of translation debt, making the system appear healthier than it is.",
        "confidence": 0.8,
        "improvement": 0.0,
        "consequence": 0.7,
        "evidence_sprints": [
          14
        ],
        "sprints_in_phase": 3
      },
      "A-S14-P06": {
        "phase": "replicable",
        "type": "friction",
        "description": "Several agents (Diana, Pat, Sanjay, Tommy, Nick) have ai_trust=0.00 and are not using AI at all, despite the system being available. Pat has 168 decisions with +13/-1 AI experiences but trust=0.00, and Sanjay has 144 decisions with +22/-3 but trust=0.00. This is not a lack of exposure but a deliberate rejection of AI recommendations, even when the AI has been mostly correct.",
        "hypothesis": "These agents may have a professional identity that values human judgment over automation, or they may have had a negative experience with AI in a previous system. The positive AI experiences are not enough to overcome this bias, possibly because the AI's recommendations are not transparent enough or the agents don't understand how the AI arrived at its conclusions.",
        "confidence": 0.85,
        "improvement": 0.0,
        "consequence": 0.8,
        "evidence_sprints": [
          14
        ],
        "sprints_in_phase": 1
      },
      "A-S14-P07": {
        "phase": "theoretical",
        "type": "friction",
        "description": "Agents are engaging in boundary work by manually reviewing AI recommendations and adding their own judgment. For example, Jordan (decision #16) accepted an AI recommendation for a simple claim, but the rationale shows he verified the claim attributes before accepting. Similarly, Pat (decision #2) reviewed policy details, claim details, and damage estimate before approving, even though the AI may have provided a recommendation. This is healthy boundary work, but it adds cycle time.",
        "hypothesis": "Agents are maintaining professional accountability by not blindly accepting AI recommendations. They are performing verification steps to ensure the AI's output is correct, which is good for quality but adds time to each decision.",
        "confidence": 0.7,
        "improvement": 0.0,
        "consequence": 0.5,
        "evidence_sprints": [
          14
        ],
        "sprints_in_phase": 3
      },
      "A-S15-P00": {
        "phase": "applicable",
        "type": "friction",
        "description": "Translation debt index has risen sharply to 2.67 (from ~1.0 in prior sprints), with a corresponding spike in exception rate (5.33%) and handoff failure rate (16.0%). Decision #11 shows a concrete failure: 'AI processed settlement_ai but its output lost meaning at the handoff \u2014 downstre...' This indicates the AI-to-human handoff is degrading, likely due to the AI pipeline producing outputs that don't map cleanly to human workflow expectations.",
        "hypothesis": "The AI pipeline is auto-processing more complex claims (not just simple ones), and its output format/context isn't being translated into the human-readable format that downstream agents expect. The 'translation=True' flag in decision #11 confirms this is a semantic loss, not just a technical error.",
        "confidence": 0.92,
        "improvement": 0.0,
        "consequence": 0.95,
        "evidence_sprints": [
          15
        ],
        "sprints_in_phase": 1
      },
      "A-S15-P01": {
        "phase": "applicable",
        "type": "friction",
        "description": "Diana, the most experienced agent (553 decisions), has ai_trust=0.00 and exhaustion=8.0. She is making decisions without AI assistance (all her decisions show AI used=False). Her stress is 0.00, which is suspicious \u2014 it suggests she's disengaged or has mentally checked out. She's the only agent with exhaustion above 7.0 who also has zero AI trust.",
        "hypothesis": "Diana has been forced to manually process high volumes of claims because the AI pipeline fails on complex cases. Her exhaustion is high, but her stress is 0.00 \u2014 this is a classic 'learned helplessness' pattern where she's stopped caring about outcomes and is just going through the motions. Her zero trust is rational given the AI's poor performance on complex claims, but her disengagement is dangerous.",
        "confidence": 0.88,
        "improvement": 0.0,
        "consequence": 0.9,
        "evidence_sprints": [
          15
        ],
        "sprints_in_phase": 1
      },
      "A-S15-P02": {
        "phase": "applicable",
        "type": "friction",
        "description": "A new trust paradox emerges: agents with the MOST positive AI experiences (Pat: +15/-0, Jordan: +15/-0, Nick: +15/-0, Greg: +14/-0) have either zero trust (Pat, Nick) or very high trust (Jordan, Greg). Meanwhile, agents with mixed experiences (Sanjay: +26/-4, Alicia: +26/-2) have moderate-to-high trust. This suggests trust isn't simply a function of positive experience ratio \u2014 it's about whether the agent has had ANY negative experience to calibrate against.",
        "hypothesis": "Trust formation isn't just about the ratio of positive to negative experiences \u2014 it's about whether the agent has had a NEGATIVE experience that was significant enough to create a 'trust anchor'. Pat and Nick have never seen AI fail, so they don't trust it because they don't understand its limits. Sanjay has seen 4 failures, which were enough to destroy his trust entirely. Jordan and Greg have seen only successes and have developed an over-trust that could be dangerous.",
        "confidence": 0.85,
        "improvement": 0.0,
        "consequence": 0.85,
        "evidence_sprints": [
          15
        ],
        "sprints_in_phase": 1
      },
      "A-S15-P03": {
        "phase": "applicable",
        "type": "friction",
        "description": "The exception rate has risen to 5.33%, and critically, the exceptions are now occurring at AI-to-human handoffs (translation=True) rather than at human decision points. Decision #11 shows an exception being raised because the AI output 'lost meaning at the handoff'. This is a NEW type of exception \u2014 it's not about the AI being wrong, it's about the AI being incomprehensible to humans.",
        "hypothesis": "As the AI pipeline handles more complex claims, its outputs are becoming more nuanced and context-dependent. The current handoff format (simple text output) isn't sufficient to convey this nuance. The AI is producing correct decisions, but the 'why' is being lost, causing downstream agents to flag exceptions because they can't verify the AI's reasoning.",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.88,
        "evidence_sprints": [
          15
        ],
        "sprints_in_phase": 1
      },
      "A-S15-P04": {
        "phase": "applicable",
        "type": "friction",
        "description": "Work is piling up on Diana (553 decisions, exhaustion=8.0) while other agents like Ron (0 decisions), Tricia (0 decisions), Mike (0 decisions), and Leslie (0 decisions) are completely idle. This is a severe workload imbalance \u2014 Diana is processing 553 decisions while 4 agents process zero. The bottleneck has migrated from the AI pipeline to a single human agent.",
        "hypothesis": "The AI pipeline is auto-processing simple claims, but complex claims are being routed to Diana because she's the 'senior' adjuster. However, the routing logic isn't distributing work to other available agents. The idle agents (Ron, Tricia, Mike, Leslie) may not have the necessary skills or permissions to handle complex claims, creating a single point of failure.",
        "confidence": 0.95,
        "improvement": 0.0,
        "consequence": 0.93,
        "evidence_sprints": [
          15
        ],
        "sprints_in_phase": 1
      },
      "A-S15-P05": {
        "phase": "applicable",
        "type": "friction",
        "description": "There's a clear pattern of adoption resistance among agents with high decision counts: Diana (553, trust=0.00), Tommy (606, trust=0.11), Nick (491, trust=0.00), and Pat (178, trust=0.00). These agents have the most experience with the system but have the lowest trust. In contrast, agents with fewer decisions (Jordan: 375, trust=1.00; Alicia: 702, trust=0.97; Greg: 329, trust=1.00) have high trust. This suggests that experience with the system is inversely correlated with trust.",
        "hypothesis": "Agents with high decision counts have seen more AI failures over time (even if the failure rate is low, they've encountered more absolute failures). They've also developed their own heuristics and are more confident in their own judgment. The AI's 'black box' nature is more frustrating to them because they can't verify its reasoning as easily as they can verify their own.",
        "confidence": 0.87,
        "improvement": 0.0,
        "consequence": 0.82,
        "evidence_sprints": [
          15
        ],
        "sprints_in_phase": 1
      },
      "A-S15-P06": {
        "phase": "applicable",
        "type": "friction",
        "description": "A paradoxical pattern emerges where agents with the most negative AI experiences (Tommy: +9/-3, Sanjay: +26/-4) have LOW trust (0.11 and 0.00 respectively), while agents with fewer negative experiences (Alicia: +26/-2, Jordan: +15/-0) have HIGH trust (0.97 and 1.00). This suggests that even a small number of negative experiences can destroy trust, but the threshold varies by agent. More importantly, agents with negative experiences are NOT sharing their concerns with others \u2014 there's no evidence of informal knowledge sharing about AI failures.",
        "hypothesis": "The negative experiences are not being socialized \u2014 agents who have seen AI failures are keeping them to themselves. This means the organization isn't learning from these failures, and other agents are developing over-trust. The failure rate threshold for trust destruction appears to be around 10-15%, but this varies by agent personality and prior experience.",
        "confidence": 0.84,
        "improvement": 0.0,
        "consequence": 0.86,
        "evidence_sprints": [
          15
        ],
        "sprints_in_phase": 1
      },
      "A-S15-P07": {
        "phase": "applicable",
        "type": "friction",
        "description": "Cost per claim is $267.20, which is HIGHER than expected given the high AI adoption rate. The AI pipeline is auto-processing many claims (decisions #1, #2, #3, #13, #14, #16), which should reduce costs. However, the high handoff failure rate (16%) and exception rate (5.33%) are likely causing rework, which is driving costs up. The cost metric is inverting \u2014 AI is supposed to reduce costs, but it's actually increasing them due to translation failures.",
        "hypothesis": "The AI is correctly processing simple claims, but the handoff failures are causing downstream agents to re-process claims that were already 'completed' by AI. This double-processing is negating the cost savings from AI automation. The translation debt is the root cause \u2014 AI outputs aren't usable by humans without additional work.",
        "confidence": 0.89,
        "improvement": 0.0,
        "consequence": 0.9,
        "evidence_sprints": [
          15
        ],
        "sprints_in_phase": 1
      },
      "A-S15-P08": {
        "phase": "theoretical",
        "type": "friction",
        "description": "Agents are engaging in boundary work by manually reviewing AI recommendations and adding their own judgment, even when the AI is correct. For example, Jordan's decision #0 shows 'accept_ai' with 'AI used=True, correct=True', but the description says 'The AI recommendation aligns with the standard routing for this claim' \u2014 suggesting Jordan is still verifying the AI's work. This is healthy boundary work, but it's adding to workload and may be contributing to the bottleneck.",
        "hypothesis": "Agents are developing a healthy skepticism of AI and are verifying its outputs. This is good for accuracy but bad for efficiency. The organization needs to find a balance between verification and trust \u2014 agents should verify AI outputs for complex claims but trust them for simple claims.",
        "confidence": 0.78,
        "improvement": 0.0,
        "consequence": 0.75,
        "evidence_sprints": [
          15
        ],
        "sprints_in_phase": 2
      },
      "A-S15-P09": {
        "phase": "applicable",
        "type": "friction",
        "description": "A paradoxical pattern emerges where agents with the most positive AI experiences (Pat: +15/-0, Jordan: +15/-0, Nick: +15/-0, Greg: +14/-0) have either zero trust (Pat, Nick) or very high trust (Jordan, Greg). This suggests that trust isn't simply a function of positive experience ratio \u2014 it's about whether the agent has had ANY negative experience to calibrate against. Agents with 100% positive experiences are either completely trusting (dangerous) or completely distrusting (inefficient).",
        "hypothesis": "Trust formation isn't just about the ratio of positive to negative experiences \u2014 it's about whether the agent has had a NEGATIVE experience that was significant enough to create a 'trust anchor'. Pat and Nick have never seen AI fail, so they don't trust it because they don't understand its limits. Jordan and Greg have seen only successes and have developed an over-trust that could be dangerous.",
        "confidence": 0.83,
        "improvement": 0.0,
        "consequence": 0.85,
        "evidence_sprints": [
          15
        ],
        "sprints_in_phase": 1
      },
      "A-S16-P01": {
        "phase": "theoretical",
        "type": "friction",
        "description": "Diana, the most experienced agent (585 decisions, exhaustion=8.0), has zero AI trust and never uses AI. Pat (189 decisions) and Sanjay (163 decisions) also have zero AI trust despite having positive AI experiences (+16/-1 and +29/-2 respectively). This suggests that high-experience agents are actively rejecting AI despite evidence it works.",
        "hypothesis": "These agents have deep domain expertise and likely perceive AI as a threat to their professional judgment. Their positive AI experiences (Pat: +16/-1) are being discounted because they see AI as a 'black box' that doesn't account for the nuanced judgment they've developed over years. Diana's exhaustion (8.0) suggests she's overworked and may see AI as adding cognitive load rather than reducing it.",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.75,
        "evidence_sprints": [
          16
        ],
        "sprints_in_phase": 1
      },
      "A-S16-P02": {
        "phase": "theoretical",
        "type": "friction",
        "description": "There's a clear positive trust cascade among agents with high AI exposure. Jordan (400 decisions, +16/-0, trust=1.00), Greg (345 decisions, +11/-0, trust=1.00), and Alicia (743 decisions, +18/-6, trust=0.70) all show high trust. These agents are actively using AI (Jordan and Greg have perfect trust) and their positive experiences are reinforcing adoption.",
        "hypothesis": "These agents have had overwhelmingly positive AI experiences (zero or minimal errors) and have learned to trust the AI's judgment. Their low stress and exhaustion levels suggest that AI adoption is reducing their cognitive load, creating a virtuous cycle. Jordan and Greg's perfect trust indicates they've internalized AI as a reliable partner.",
        "confidence": 0.85,
        "improvement": 0.0,
        "consequence": 0.6,
        "evidence_sprints": [
          16
        ],
        "sprints_in_phase": 1
      },
      "A-S16-P03": {
        "phase": "theoretical",
        "type": "friction",
        "description": "Agents are engaging in boundary work by manually reviewing AI recommendations even when they approve them. Decision samples show humans like Diana, Pat, and Tommy providing detailed justifications for their approvals, suggesting they're not simply rubber-stamping AI outputs but actively verifying them.",
        "hypothesis": "Agents are maintaining professional identity by demonstrating they're adding value beyond AI. They're also hedging against potential AI errors by documenting their own review process. This is healthy boundary work that maintains accountability, but it adds time and cost.",
        "confidence": 0.7,
        "improvement": 0.0,
        "consequence": 0.5,
        "evidence_sprints": [
          16
        ],
        "sprints_in_phase": 1
      },
      "A-S16-P04": {
        "phase": "theoretical",
        "type": "friction",
        "description": "Tommy, with the most negative AI experience (+24/-9), has zero AI trust despite having 24 positive experiences. Nick (+13/-2) has near-zero trust (0.03). This is paradoxical because the positive experiences should outweigh the negative ones, yet trust remains at zero.",
        "hypothesis": "These agents are likely experiencing 'negativity bias' where the 9 errors Tommy saw are more salient than the 24 successes. The errors may have been in high-stakes situations (e.g., complex claims) where the cost of error was high. Nick's exhaustion (9.0) suggests he's overworked and may not have the cognitive capacity to properly evaluate AI performance.",
        "confidence": 0.8,
        "improvement": 0.0,
        "consequence": 0.7,
        "evidence_sprints": [
          16
        ],
        "sprints_in_phase": 1
      },
      "A-S16-P05": {
        "phase": "empirical",
        "type": "friction",
        "description": "Cost per claim is $258.48, which is HIGHER than expected given the high AI adoption rate. The AI pipeline is auto-processing many simple claims (decisions 5, 8, 9, 13, 14, 19), yet costs remain high. This suggests that the AI processing isn't actually reducing costs as intended.",
        "hypothesis": "The high translation debt is likely causing rework that offsets AI efficiency gains. When AI auto-processes a claim, it may not capture all context, leading to downstream exceptions and manual interventions. The 12% handoff failure rate suggests that AI-processed claims are being sent back for clarification, negating the cost savings.",
        "confidence": 0.75,
        "improvement": 0.0,
        "consequence": 0.85,
        "evidence_sprints": [
          16
        ],
        "sprints_in_phase": 1
      },
      "A-S16-P06": {
        "phase": "theoretical",
        "type": "friction",
        "description": "There's a surprising disconnect between exhaustion and AI trust. Diana has exhaustion=8.0 with trust=0.00, but Mike has exhaustion=8.0 with trust=0.73. Nick has exhaustion=9.0 with trust=0.03. This suggests that exhaustion doesn't consistently predict AI trust, and the relationship is more complex than expected.",
        "hypothesis": "Exhaustion may interact with other factors like experience level and AI exposure. Diana's exhaustion likely comes from manually processing complex claims, making her resent AI. Mike's exhaustion may come from other factors (e.g., workload outside this system), and he sees AI as a relief. Nick's exhaustion may be causing cognitive fatigue that makes him less able to trust AI, or his negative experiences are amplified by exhaustion.",
        "confidence": 0.65,
        "improvement": 0.0,
        "consequence": 0.7,
        "evidence_sprints": [
          16
        ],
        "sprints_in_phase": 1
      },
      "A-S16-P07": {
        "phase": "theoretical",
        "type": "friction",
        "description": "There's a paradoxical relationship between decision volume and AI trust. Rachel has 223 decisions but zero AI trust and zero AI experience. Tommy has 649 decisions with zero trust. However, Jordan has 400 decisions with perfect trust. This suggests that decision volume alone doesn't predict trust, and some agents are making many decisions without ever engaging with AI.",
        "hypothesis": "The system may be routing certain agents to manual processing regardless of AI availability. Rachel's zero AI experience suggests she's never been offered AI recommendations. Tommy's negative experiences may have caused him to opt out of AI. Jordan may have been in a role where AI was more readily available or better integrated.",
        "confidence": 0.6,
        "improvement": 0.0,
        "consequence": 0.55,
        "evidence_sprints": [
          16
        ],
        "sprints_in_phase": 1
      }
    }
  },
  "track_b": {
    "phase_summary": {
      "theoretical": 0,
      "empirical": 0,
      "applicable": 2,
      "replicable": 8,
      "impact": 2
    },
    "by_type": {
      "friction": 12,
      "absence": 0
    },
    "discoveries": {
      "AGENT-GAP-B-fnol_intake": {
        "phase": "replicable",
        "type": "friction",
        "description": "Agents surfaced missing handoff context at fnol_intake (4 decisions)",
        "hypothesis": "If we standardize the handoff into fnol_intake, agents will stop surfacing missing context",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.5,
        "evidence_sprints": [
          1,
          2,
          3,
          4,
          5,
          6,
          7,
          8,
          9,
          10,
          11,
          12,
          13,
          14,
          15,
          16
        ],
        "sprints_in_phase": 8
      },
      "AGENT-GAP-B-investigation_human": {
        "phase": "replicable",
        "type": "friction",
        "description": "Agents surfaced missing handoff context at investigation_human (4 decisions)",
        "hypothesis": "If we standardize the handoff into investigation_human, agents will stop surfacing missing context",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.5,
        "evidence_sprints": [
          1,
          2,
          3,
          4,
          5,
          6,
          7,
          8,
          9,
          10,
          11,
          12,
          13,
          14,
          15,
          16
        ],
        "sprints_in_phase": 7
      },
      "AGENT-GAP-B-damage_estimation_human": {
        "phase": "replicable",
        "type": "friction",
        "description": "Agents surfaced missing handoff context at damage_estimation_human (4 decisions)",
        "hypothesis": "If we standardize the handoff into damage_estimation_human, agents will stop surfacing missing context",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.1111111111111111,
        "evidence_sprints": [
          1,
          2
        ],
        "sprints_in_phase": 1
      },
      "AGENT-GAP-B-coverage_verification_human": {
        "phase": "replicable",
        "type": "friction",
        "description": "Agents surfaced missing handoff context at coverage_verification_human (5 decisions)",
        "hypothesis": "If we standardize the handoff into coverage_verification_human, agents will stop surfacing missing context",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.5555555555555556,
        "evidence_sprints": [
          1,
          2,
          3,
          4,
          5,
          6,
          7,
          8,
          9,
          10,
          11,
          12,
          13,
          14,
          15,
          16
        ],
        "sprints_in_phase": 7
      },
      "AGENT-GAP-B-liability_determination_human": {
        "phase": "impact",
        "type": "friction",
        "description": "Agents surfaced missing handoff context at liability_determination_human (5 decisions)",
        "hypothesis": "If we standardize the handoff into liability_determination_human, agents will stop surfacing missing context",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.6666666666666666,
        "evidence_sprints": [
          1,
          2,
          3,
          4,
          5,
          6,
          7,
          8,
          9,
          10,
          11,
          12,
          13,
          14,
          15,
          16
        ],
        "sprints_in_phase": 4
      },
      "AGENT-GAP-B-settlement_human": {
        "phase": "impact",
        "type": "friction",
        "description": "Agents surfaced missing handoff context at settlement_human (6 decisions)",
        "hypothesis": "If we standardize the handoff into settlement_human, agents will stop surfacing missing context",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.6153846153846154,
        "evidence_sprints": [
          1,
          2,
          3,
          4,
          5,
          6,
          7,
          8,
          9,
          10,
          11,
          12,
          13,
          14,
          15,
          16
        ],
        "sprints_in_phase": 4
      },
      "AGENT-GAP-B-approval": {
        "phase": "replicable",
        "type": "friction",
        "description": "Agents surfaced missing handoff context at approval (6 decisions)",
        "hypothesis": "If we standardize the handoff into approval, agents will stop surfacing missing context",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.32,
        "evidence_sprints": [
          1,
          2,
          3,
          4,
          5,
          6,
          7,
          8,
          9,
          10,
          11,
          12,
          13,
          14,
          15,
          16
        ],
        "sprints_in_phase": 2
      },
      "AGENT-GAP-B-payment": {
        "phase": "replicable",
        "type": "friction",
        "description": "Agents surfaced missing handoff context at payment (8 decisions)",
        "hypothesis": "If we standardize the handoff into payment, agents will stop surfacing missing context",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.28,
        "evidence_sprints": [
          1,
          2
        ],
        "sprints_in_phase": 1
      },
      "AGENT-GAP-B-subrogation": {
        "phase": "replicable",
        "type": "friction",
        "description": "Agents surfaced missing handoff context at subrogation (8 decisions)",
        "hypothesis": "If we standardize the handoff into subrogation, agents will stop surfacing missing context",
        "confidence": 0.9,
        "improvement": 0.0,
        "consequence": 0.4,
        "evidence_sprints": [
          1,
          2,
          3,
          4,
          5,
          6,
          7,
          8,
          9,
          10,
          11,
          12,
          13,
          14,
          15,
          16
        ],
        "sprints_in_phase": 3
      },
      "AGENT-GAP-B-escalated_to_kathryn": {
        "phase": "replicable",
        "type": "friction",
        "description": "Agents surfaced missing handoff context at escalated_to_kathryn (1 decisions)",
        "hypothesis": "If we standardize the handoff into escalated_to_kathryn, agents will stop surfacing missing context",
        "confidence": 0.5,
        "improvement": 0.0,
        "consequence": 1.0,
        "evidence_sprints": [
          2,
          3,
          4,
          5,
          6,
          9,
          15
        ],
        "sprints_in_phase": 2
      },
      "AGENT-GAP-B-coverage_verification_ai": {
        "phase": "applicable",
        "type": "friction",
        "description": "Agents surfaced missing handoff context at coverage_verification_ai (1 decisions)",
        "hypothesis": "If we standardize the handoff into coverage_verification_ai, agents will stop surfacing missing context",
        "confidence": 0.5,
        "improvement": 0.0,
        "consequence": 0.0625,
        "evidence_sprints": [
          5
        ],
        "sprints_in_phase": 3
      },
      "AGENT-GAP-B-settlement_ai": {
        "phase": "applicable",
        "type": "friction",
        "description": "Agents surfaced missing handoff context at settlement_ai (1 decisions)",
        "hypothesis": "If we standardize the handoff into settlement_ai, agents will stop surfacing missing context",
        "confidence": 0.5,
        "improvement": 0.0,
        "consequence": 0.06666666666666667,
        "evidence_sprints": [
          5
        ],
        "sprints_in_phase": 3
      }
    }
  }
}