{
  "schema": "memcat-local-llm-independent-internal-audit-v1",
  "completedAt": "2026-10-05T22:50:39.604810+00:00",
  "reviewerScope": "Separate internal agent performed read-only audit of frozen experiment sources and actual receipts. This is not unaffiliated external validation. No inference, model changes, source changes or scored retries were performed during this audit.",
  "source": {
    "analysisPath": "docs/memcat03/runs/22-local-llm-primary-analysis.json",
    "analysisSha256": "582a276a3995f5c10be8553b632fe60a9b82462a108fdaeb0bde82f5c5dbd858",
    "planPath": "docs/memcat03/runs/20-local-llm-comparison-plan.json",
    "planSha256": "6902a2046db81a8990136e50da5da40b1c788829d4a1286f1ba6dead5db2509d",
    "runDirectory": "docs/memcat03/runs/local-llm-primary-001",
    "checkedCollectionSources": {
      "packages/sdk/src/classifier.ts": "72fbe006725f92b0d78f719df67b0ab96080f80f4ee1ed937debf10471071372",
      "packages/sdk/src/evaluation.ts": "187e95fdccf0d82e3b43794aab26c0744db42aad02fcc1f1b6bcb8dd8f8525c4",
      "scripts/memcat-encoder.ts": "5ebf84c7024dafd0da5768b0f4677cfa62f7409d2cc7ab9575f909a9f9a59b97",
      "scripts/compare-memcat-local-llm.ts": "ddba437172173fefad7de04bcf575d7ef6a997856297db7e4d46c7537703df6a",
      "package-lock.json": "d4d654838c6cc1248b0c05ad12ea33efb656d2e9f32fcc7bb3cf46ad5c44d2f6",
      "docs/memcat03/runs/20a-local-llm-denominators-and-stress.json": "62ff33736ada3867a8dda207be7945d49caa53b5385dc52b3b3f1710cc61fc42"
    }
  },
  "integrity": {
    "passed": true,
    "startAndResultPairs": 1540,
    "scoredCaseCountPerArm": 770,
    "uniqueRawScoredLlmResponses": 997,
    "segments": 1,
    "resumed": false,
    "postflightArms": [
      "direct",
      "hybrid"
    ],
    "terminalComplete": true,
    "terminalMatchesImmutableRecords": true,
    "analysisMatchesImmutableRecords": true,
    "modelAndOllamaIdentityVerified": true,
    "reportedCountersAndDurationsRecomputedFromRaw": true,
    "noTruncationOrProtocolFailures": true,
    "noObservedSourceIdentityDrift": true,
    "originalMemcatPredictionsIdentical": 770,
    "maxOriginalScoreOrMarginDrift": 0,
    "errorReceiptFiles": 0,
    "unknownLlmAttempts": 0,
    "warmupsPerArm": 3,
    "coldCacheWarmupsHaveZeroCachedPromptTokens": true,
    "sameWarmupInputsAndRequestHashes": true,
    "noScoredOutputsReused": "Runner source issues actual requests for direct and fallback. All 997 response bodies have unique hashes/timestamps; fallback request hashes intentionally match the corresponding direct requests."
  },
  "recomputed": {
    "direct": {
      "planned": 770,
      "recorded": 770,
      "correct": 535,
      "wrong": 218,
      "review": 17,
      "operationalFailures": 0,
      "allInputAccuracy": 0.6948051948051948,
      "acceptedPrecision": 0.7104913678618858,
      "macroF1": 0.6928925178804731,
      "llmCalls": 770,
      "promptTokens": 2097089,
      "outputTokens": 6842,
      "cachedPromptTokens": 2082923,
      "promptTokenRange": [
        2714,
        2782
      ],
      "scoredTotalMs": 299674.2724570005,
      "scoredP50Ms": 353.5702080000192,
      "scoredP95Ms": 794.0454169999866
    },
    "hybrid": {
      "planned": 770,
      "recorded": 770,
      "correct": 642,
      "wrong": 123,
      "review": 5,
      "operationalFailures": 0,
      "allInputAccuracy": 0.8337662337662337,
      "acceptedPrecision": 0.8392156862745098,
      "macroF1": 0.8376263535369654,
      "llmCalls": 227,
      "promptTokens": 618519,
      "outputTokens": 2022,
      "cachedPromptTokens": 614063,
      "promptTokenRange": [
        2716,
        2762
      ],
      "scoredTotalMs": 87717.26990999957,
      "scoredP50Ms": 8.40562500001397,
      "scoredP95Ms": 436.4109579999931
    }
  },
  "paired": {
    "bothCorrect": 529,
    "hybridOnlyCorrect": 113,
    "directOnlyCorrect": 6,
    "neitherCorrect": 122,
    "accuracyDifference": 0.13896103896103895,
    "oneSided95Lower": 0.11948051948051948,
    "replicates": 20000,
    "seed": "0x4d434c31",
    "method": "Independent JavaScript implementation reproduced the stratified paired bootstrap exactly: sorted 77 labels, ten draws per stratum, carried xorshift32, nearest-rank 5th percentile. Repeated calls were not pooled as independent rows."
  },
  "reductions": {
    "actualScoredLlmCalls": 0.7051948051948052,
    "reportedPromptPlusOutputTokens": 0.7050563920584848,
    "scoredClientWallTime": 0.7072912893362046
  },
  "failureConcentration": {
    "localAccepted": 543,
    "localAcceptedCorrect": 527,
    "localAcceptedWrong": 16,
    "localAcceptedPrecision": 0.9705340699815838,
    "localReviewsSentToActualFallback": 227,
    "localOperationalFailures": 0,
    "fallbackCorrect": 115,
    "fallbackWrong": 107,
    "fallbackReview": 5,
    "fallbackAllInputAccuracy": 0.5066079295154186,
    "fallbackLabelsIdenticalToCorrespondingDirectLabels": 227,
    "hybridLossesAgainstCorrectDirect": 6,
    "allSixHybridLossesComeFromLocalAcceptance": true,
    "hybridGainsAgainstIncorrectDirect": 113,
    "interpretation": "The combined route improves this baseline by replacing many wrong Qwen labels with correct local labels. It still misses the 90% target because the fallback answers only 115 of 227 deferred cases correctly, alongside 16 incorrect local acceptances. Review and wrong outputs are distinct; there were zero operational failures."
  },
  "diagnosticExamples": [
    {
      "id": "banking77-test-01852",
      "text": "How long can an EU transfer take?",
      "expectedLabel": "pending_transfer",
      "directLabel": "pending_transfer",
      "hybridLabel": "transfer_timing",
      "route": "local",
      "localScore": 0.907779664166621,
      "localMargin": 0.11671811534680931
    },
    {
      "id": "banking77-test-01860",
      "text": "When will the transfer go through?",
      "expectedLabel": "pending_transfer",
      "directLabel": "pending_transfer",
      "hybridLabel": "balance_not_updated_after_bank_transfer",
      "route": "local",
      "localScore": 0.9138251594578045,
      "localMargin": 0.0975235345409291
    },
    {
      "id": "banking77-test-02716",
      "text": "Where is my money? I transfered it and it isn't in my account.",
      "expectedLabel": "balance_not_updated_after_bank_transfer",
      "directLabel": "balance_not_updated_after_bank_transfer",
      "hybridLabel": "transfer_timing",
      "route": "local",
      "localScore": 0.874024829264662,
      "localMargin": 0.13285667532962608
    },
    {
      "id": "banking77-test-02064",
      "text": "How long does it take for a transfer?",
      "expectedLabel": "transfer_timing",
      "directLabel": "transfer_timing",
      "hybridLabel": "balance_not_updated_after_bank_transfer",
      "route": "local",
      "localScore": 0.9368515469591228,
      "localMargin": 0.09609038504308443
    },
    {
      "id": "banking77-test-01392",
      "text": "How many virtual cards do I get?",
      "expectedLabel": "disposable_card_limits",
      "directLabel": "disposable_card_limits",
      "hybridLabel": "getting_virtual_card",
      "route": "local",
      "localScore": 0.8794904911312043,
      "localMargin": 0.1300754467554741
    },
    {
      "id": "banking77-test-03036",
      "text": "Is there any way to verify who I am?",
      "expectedLabel": "verify_my_identity",
      "directLabel": "verify_my_identity",
      "hybridLabel": "why_verify_identity",
      "route": "local",
      "localScore": 0.875051450664003,
      "localMargin": 0.09543789781246048
    }
  ],
  "baselineLimitations": [
    "Frozen baseline uses one hash-selected training exemplar per label and humanized label identifiers, whereas MemCat uses 32 examples plus a description. This is a reproducible reference configuration, not evidence against the best configured LLM classifier.",
    "One concrete taxonomy mismatch: official get_physical_card is described literally as get physical card, but its selected training exemplar asks about the card PIN. Nine of ten scored rows in that class were predicted change_pin, and none were correct. This supports investigating taxonomy definitions on training/development data; it does not prove a causal fix or justify editing this test.",
    "The other major observed confusions include card-payment versus cash-withdrawal exchange rates, reasons for identity checks versus how to verify identity, and transfer timing/pending/missing-balance labels. These are error descriptions, not a new protected evaluation.",
    "Direct then hybrid ran once on the same machine with normal prefix caching. Cold Qwen warmup was about 4.10 seconds before direct and 10.84 seconds before hybrid, illustrating runtime variability. The scored aggregate excludes setup/warmups/receipt I/O as predeclared; it is not a universal throughput claim.",
    "Approximately 99% of reported prompt tokens were marked cached in both arms. Total reported tokens are not uncached model computation, paid billed tokens or total compute/energy.",
    "The 770 cases are a reused public benchmark subset with potential pretraining exposure. They do not establish customer outcomes, multilingual performance or near-domain unknown robustness.",
    "The separate 14-case authored stress suite was not rerun here. Its previously observed mixed-intent wrong acceptance remains unresolved and visible."
  ],
  "gates": {
    "confirmed": {
      "allPlannedReceipts": true,
      "hybridAccuracyAtLeast90Percent": false,
      "pairedLowerBoundAtLeastMinus2Points": true,
      "finalHybridFailuresAtMost1Percent": true,
      "atLeast50PercentFewerLlmCalls": true,
      "atLeast50PercentFewerReportedTokens": true,
      "atLeast30PercentLowerScoredTime": true
    },
    "passed": false,
    "failedGate": "hybridAccuracyAtLeast90Percent",
    "observedHybridAccuracy": 0.8337662337662337,
    "requiredHybridAccuracy": 0.9,
    "headline": "Fewer calls and higher accuracy than this local reference baseline; the combined system still failed the predeclared 90% accuracy target."
  },
  "fixedRubricAssessment": {
    "rubricPath": "docs/MEMCAT_03_EVALUATION_PLAN.md",
    "subjectiveInternalAssessment": true,
    "dimensions": [
      {
        "dimension": "usefulClassification",
        "weight": 0.25,
        "score": 6.5,
        "reason": "Protected public selective result remains useful; mixed-intent gate fails, and combined reference workflow achieves only 83.38% all-input accuracy. No representative customer evidence."
      },
      {
        "dimension": "reliabilityAndControl",
        "weight": 0.2,
        "score": 8.5,
        "reason": "Reviewed bounded SDK contracts, actual clean consumer verification, transparent failures, immutable run receipts and independent internal reconciliation. This is engineering evidence, not external certification."
      },
      {
        "dimension": "developerExperience",
        "weight": 0.15,
        "score": 7,
        "reason": "Working semantic starter, package examples, portable index, local evaluation and replay. Independent integrations without author assistance are unmeasured."
      },
      {
        "dimension": "measuredEconomics",
        "weight": 0.2,
        "score": 5,
        "reason": "Actual local baseline and hybrid requests show lower request count and measured time with higher accuracy, but the predeclared absolute quality floor fails. Paid costs, full integration/review burden and representative workflow ROI remain unknown."
      },
      {
        "dimension": "independentUsefulnessAndRetention",
        "weight": 0.2,
        "score": 2,
        "reason": "Pilot plan and tooling exist; no unaffiliated recurring users or retention evidence. No credit for invented external validation."
      }
    ],
    "weightedOverall": 5.775,
    "roundedOverall": 5.8,
    "engineeringOnlyApproximate": 8,
    "engineeringScoreScope": "Technical alpha: runtime controls, package integration and auditability. This cannot substitute for the overall product score.",
    "overallTarget": 7.5,
    "overallTargetDemonstrated": false,
    "scoreChangeReason": "The real local comparison adds partial economics evidence beyond fixture tests, while its failed absolute quality target and missing independent workflows prevent a passing product score."
  },
  "nextMinimumEvidence": [
    "Preserve this failed comparison and all original thresholds. Do not retune on its 770 scored rows or relabel a rerun as new held-out evidence.",
    "Improve and select the fallback configuration on training/development data: audited category definitions and example budget, explicit confusion-pair distinctions, and a measured choice of model. Freeze choices before a new protected evaluation; do not assume a larger model will pass.",
    "Obtain an authorized real workflow with independently labelled unseen traffic and its existing baseline. Evaluate total correct outcomes, harmful wrong routes, review work, actual paid usage when applicable, setup, retries and end-to-end latency.",
    "Resolve mixed-intent and near-domain unknown rejection on development data, then test a fresh independently specified stress/unknown set. Preserve existing failures as regressions.",
    "Have unaffiliated developers install and use the SDK without author assistance. The fixed rubric requires three real trials and at least two continuing after two weeks before strong independent-usefulness credit.",
    "If a general speed claim matters, predeclare balanced-order repeated performance runs under stated load/cache conditions. Keep repeats separate from independent accuracy sample size."
  ]
}
