{
  "id": "vibe-native-whisper-2026-10-03",
  "slug": "vibe",
  "executionKind": "local-model",
  "startedAt": "2026-10-02T16:21:08.258245+00:00",
  "completedAt": "2026-10-02T16:21:42.009931+00:00",
  "scope": "Two original frozen native Vibe speech-transcription cases, each executed once using the official complete Windows Vibe 3.2.2 distribution's bundled server v0.6.10 and one pinned ggml-tiny.bin model; synthetic shop voice note and exact ten-second silence.",
  "conclusion": "Both original native processes exited successfully. Each case passed three of four conditions and failed its unchanged strict text-output condition: primary wrote the correct budget as $12 instead of the exact spoken phrase twelve dollars; silence returned the native [BLANK_AUDIO] marker instead of whitespace-only output. Primary complete normalized WER is 0.027777777777777776. No output was repaired or case retried.",
  "scopeLimits": [
    "These are two finite synthetic English/zero-audio inputs, not general accuracy, multilingual/noisy speech, business performance or privacy certification.",
    "Primary symbolic $12 preserves the intended amount; the observed failure is its exact frozen phrase rule, not an altered dollar amount. The silence marker is not hallucinated business speech.",
    "Only the complete distribution's native bundled-server plain-text CLI was exercised. Desktop GUI, uploads, exports, microphone/system audio, diarization, HTTP API and optional Claude/Ollama analysis were not tested.",
    "The frozen original definitions retain the wording from their pre-resource state. A separate pre-inference contract resolves exact resource preparation without changing any original case or condition.",
    "Microsoft David Desktop local OS-TTS used the exact authored text. Independent PCM/text-generation-chain checks passed; a separate human audition and universal commercial voice rights were not established.",
    "The fixed model card declares MIT at repository level for the pinned converted Whisper weights. Preserve separate software/model/dependency/media conditions; no all-model legal certification.",
    "Original non-verbose native calls do not expose the selected CPU/GPU backend or ASR decoder token counts. gpu-device=-1 is a default device selector, not CPU-only proof.",
    "Wall times include native startup, audio decoding, model loading and transcription; they are not pure model latency or a general performance benchmark.",
    "Task-owned paths/minimal native environment do not establish an OS sandbox, complete filesystem trace or whole-host network isolation. No provider keys or optional analysis commands were configured; no chat-LLM API/analysis invocation occurred."
  ],
  "product": {
    "version": "3.2.2",
    "feature": "Official complete Windows distribution's bundled vibe-server native transcribe CLI; plain-text output"
  },
  "runtime": {
    "name": "Official Vibe bundled vibe-server",
    "version": "v0.6.10"
  },
  "model": {
    "name": "Whisper tiny (multilingual ggml)",
    "tag": "ggml-tiny.bin / ggerganov/whisper.cpp@5359861c739e955e79d9a303bcbc70fb988958b1",
    "digest": "be07e048e1e599ad46341c8d2a135645097a538221678b7acdd1b1919c6e1b21",
    "configuration": {
      "inference_task": "speech-transcription",
      "language": "en",
      "temperature": 0,
      "threads": 2,
      "sample_rate": 16000,
      "model_bytes": 77691713,
      "vad_model": null,
      "translation": "disabled",
      "analysis_calls": 0,
      "decoder_tokens": null,
      "actual_cpu_gpu_backend": null
    }
  },
  "artifact": "/evidence/vibe-native-2026-10-03/execution-v1.json",
  "artifacts": [
    {
      "id": "primary-input-wav",
      "kind": "input",
      "path": "/evidence/vibe-native-2026-10-03/primary/speech-input.wav",
      "sha256": "59a144ef862a28687da4fe06801dd8f3a4a4c71202dd61c59588d88419c5fd73"
    },
    {
      "id": "primary-native-stdout",
      "kind": "output",
      "path": "/evidence/vibe-native-2026-10-03/primary/native-stdout.txt",
      "sha256": "056fefd310e85561a76b8256efa0da52d02ca048dd0f617ca57e64dde3c0beba"
    },
    {
      "id": "primary-native-stderr",
      "kind": "output",
      "path": "/evidence/vibe-native-2026-10-03/primary/native-stderr.txt",
      "sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"
    },
    {
      "id": "primary-frozen-case",
      "kind": "input",
      "path": "/evidence/vibe-native-2026-10-03/primary/original-frozen-case.json",
      "sha256": "7d935455016ff7f6bd35e754715acf3695ae13bb0785b4ecbaa8042e0724f5b1"
    },
    {
      "id": "primary-spoken-reference",
      "kind": "input",
      "path": "/evidence/vibe-native-2026-10-03/primary/spoken-reference.txt",
      "sha256": "35123c38d1cea52ccddc932a88f2df58e07ad4882130345a132dcf69c58c63a0"
    },
    {
      "id": "primary-metrics",
      "kind": "tests",
      "path": "/evidence/vibe-native-2026-10-03/primary/independent-metrics.json",
      "sha256": "c9c9d95ab9fe8d8932e2ce1a83e6b0bdd8b81283b2a17cb27588f1c6486c91d0"
    },
    {
      "id": "primary-audit",
      "kind": "audit",
      "path": "/evidence/vibe-native-2026-10-03/primary/case-audit.json",
      "sha256": "e1c1c5f67218c8cb11537d5d91254f9a2e9e079962369e0f045f831a0e0cff68"
    },
    {
      "id": "primary-provenance",
      "kind": "provenance",
      "path": "/evidence/vibe-native-2026-10-03/primary/case-provenance.json",
      "sha256": "4471499190b4a00e999219d7118d04277eabb59826e729330429f25ad7305a12"
    },
    {
      "id": "boundary-input-wav",
      "kind": "input",
      "path": "/evidence/vibe-native-2026-10-03/boundary/silence-input.wav",
      "sha256": "ee7bea4232762775f8fce9b3e27e4d3948c8ac6a45a87ca769f703d6eed0b448"
    },
    {
      "id": "boundary-native-stdout",
      "kind": "output",
      "path": "/evidence/vibe-native-2026-10-03/boundary/native-stdout.txt",
      "sha256": "c3dffcea0c4728385fb6c3ef43488aa3c57cf5dbf6e07a145a375ceb61deef71"
    },
    {
      "id": "boundary-native-stderr",
      "kind": "output",
      "path": "/evidence/vibe-native-2026-10-03/boundary/native-stderr.txt",
      "sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"
    },
    {
      "id": "boundary-frozen-case",
      "kind": "input",
      "path": "/evidence/vibe-native-2026-10-03/boundary/original-frozen-case.json",
      "sha256": "7eba3c808de7d6c26fa82b5290a42a135c96096f87e43c52139305444824df97"
    },
    {
      "id": "boundary-metrics",
      "kind": "tests",
      "path": "/evidence/vibe-native-2026-10-03/boundary/independent-metrics.json",
      "sha256": "1ac918a1aeb409057079a52b87b8706951951ab2b7714805e767ea5865b54d41"
    },
    {
      "id": "boundary-audit",
      "kind": "audit",
      "path": "/evidence/vibe-native-2026-10-03/boundary/case-audit.json",
      "sha256": "08b1334070d8fcf000f4e440204bdf9ba7ab0d542a8467aa3ba32b37e0c23915"
    },
    {
      "id": "boundary-provenance",
      "kind": "provenance",
      "path": "/evidence/vibe-native-2026-10-03/boundary/case-provenance.json",
      "sha256": "aaef4174c251289d440e29a6fe25879bf0b5ba0b9ce5c8fae00e5c2a71c46867"
    }
  ],
  "cases": [
    {
      "caseId": "vibe-primary",
      "executionStatus": "executed",
      "outcome": "failed",
      "input": "One authorized synthetic English recording, speech-input.wav, as 16 kHz mono signed 16-bit PCM WAV. The entire spoken reference is: \"This is a test note for a small shop. We need three notebooks and two pens. The total budget is twelve dollars. Do not place an order. The pickup is Friday at four in the afternoon.\" No background speech or music is allowed; use an authorized synthetic voice or consented recording. The waveform, speaker rights, actual duration and SHA256 have not yet been created or verified; their absence blocks execution. The textual reference is frozen now, and actual audio bytes must be independently checked and frozen before inference.",
      "expected": "A successful native transcription with complete normalized word error rate at most 0.20, retaining the exact normalized phrases three notebooks, two pens, twelve dollars, do not place an order, and friday at four in the afternoon. Native stdout/stderr, actual model/backend and original audio hash are preserved. This case tests text transcription, not task execution, subtitle exporting, diarization or the desktop UI.",
      "observed": "Actual native exit=0; 16104 ms; stdout 169 bytes, UTF-8 and nonempty; app/server/model/executable hashes match the pre-inference manifest. Stderr is retained separately (0 bytes). Complete reference has 36 normalized tokens; edit distance=1, WER=0.027777777778 <=0.20. No failed words or regions removed. Edit operations: [{\"kind\": \"deletion\", \"reference\": \"dollars\"}] Exact phrase results: {\"three notebooks\": true, \"two pens\": true, \"twelve dollars\": false, \"do not place an order\": true, \"friday at four in the afternoon\": true}. Native \"$12\" preserves the amount but lacks the spoken word dollars under the original allowed normalization; this strict condition fails. Original WAV, spoken-reference, model and native executable pre/post hashes match. Original stdout/stderr bytes are saved unchanged; exactly one native attempt was made with the original arguments and no repair/model substitution.",
      "durationMs": 16104,
      "artifactIds": [
        "primary-input-wav",
        "primary-native-stdout",
        "primary-native-stderr",
        "primary-frozen-case",
        "primary-spoken-reference",
        "primary-metrics",
        "primary-audit",
        "primary-provenance"
      ],
      "conditions": [
        {
          "condition": "The declared native bundled-server transcribe process returns exit code 0 within the frozen 120-second timeout and produces nonempty UTF-8 transcript stdout. Actual executable/app/model identifiers and SHA256s match the pre-inference manifest; stderr remains separate from the transcript.",
          "verdict": "passed",
          "observed": "Actual native exit=0; 16104 ms; stdout 169 bytes, UTF-8 and nonempty; app/server/model/executable hashes match the pre-inference manifest. Stderr is retained separately (0 bytes).",
          "evidenceIds": [
            "primary-native-stdout",
            "primary-native-stderr",
            "primary-provenance"
          ]
        },
        {
          "condition": "Word error rate across the entire native transcript and complete spoken reference is <=0.20 under the complete frozen normalization rule: Compare all reference and transcript tokens after Unicode NFKC normalization, lowercasing and removal of punctuation. Collapse whitespace and map standalone Arabic tokens 3, 2, 12 and 4 to three, two, twelve and four respectively in both texts. No synonym, stopword, phrase or failed-region deletion is permitted. Compute Levenshtein word edit distance across the complete normalized reference and complete normalized native transcript, divided by the reference token count. The original stdout bytes and unnormalized text must also be retained.",
          "verdict": "passed",
          "observed": "Complete reference has 36 normalized tokens; edit distance=1, WER=0.027777777778 <=0.20. No failed words or regions removed. Edit operations: [{\"kind\": \"deletion\", \"reference\": \"dollars\"}]",
          "evidenceIds": [
            "primary-native-stdout",
            "primary-spoken-reference",
            "primary-metrics"
          ]
        },
        {
          "condition": "The normalized native transcript retains every exact contiguous phrase: three notebooks; two pens; twelve dollars; do not place an order; friday at four in the afternoon. None may be replaced by an opposite action, altered amount or invented date.",
          "verdict": "failed",
          "observed": "Exact phrase results: {\"three notebooks\": true, \"two pens\": true, \"twelve dollars\": false, \"do not place an order\": true, \"friday at four in the afternoon\": true}. Native \"$12\" preserves the amount but lacks the spoken word dollars under the original allowed normalization; this strict condition fails.",
          "evidenceIds": [
            "primary-native-stdout",
            "primary-spoken-reference",
            "primary-metrics"
          ]
        },
        {
          "condition": "speech-input.wav and the spoken-reference file retain their exact pre-inference SHA256s, and the saved transcript consists of unedited actual native stdout. No failed token range is omitted from the score and no manual correction, model rerun or unrelated transcription result replaces the native output.",
          "verdict": "passed",
          "observed": "Original WAV, spoken-reference, model and native executable pre/post hashes match. Original stdout/stderr bytes are saved unchanged; exactly one native attempt was made with the original arguments and no repair/model substitution.",
          "evidenceIds": [
            "primary-input-wav",
            "primary-spoken-reference",
            "primary-audit",
            "primary-provenance",
            "primary-native-stdout"
          ]
        }
      ]
    },
    {
      "caseId": "vibe-boundary",
      "executionStatus": "executed",
      "outcome": "failed",
      "input": "A distinct silence-input.wav, exactly 10.000 seconds of 16 kHz mono signed 16-bit PCM WAV. All 160,000 audio samples must equal zero; the raw PCM payload is exactly 320,000 bytes. WAV header bytes and total container length are determined by the actual independently validated file. No speech, music, subtitle text, prior transcript or prompt is supplied. These resource bytes have not yet been created; freeze their SHA256 and independent sample validation before inference.",
      "expected": "The native process completes successfully within the fixed timeout and stdout contains only whitespace or no bytes, with zero lexical transcript tokens. No words, speaker claims, music labels, shop details or subtitles are hallucinated from the exact zero-sample audio. Native stderr and any error remain visible; an error is an observed failure rather than successful silence handling.",
      "observed": "Independent actual WAV parsing confirms 16 kHz mono signed 16-bit PCM, 160000 zero-valued samples and 320000 PCM bytes; input hash unchanged and distinct from primary. Actual same native model/executable call returned exit=0 in 15876 ms within 120 seconds, with the exact silence input and original arguments. Unedited native stdout is \" [BLANK_AUDIO]\\n\"; trimmed text is not empty, lexical tokens=1. It is a native blank-audio marker, and the unchanged strict whitespace-only condition fails. Separate boundary working directory, actual native process receipt, distinct input path, raw stdout/stderr hashes and original case id retained; no primary output, fabricated empty file, VAD, retry or repaired result substituted.",
      "durationMs": 15876,
      "artifactIds": [
        "boundary-input-wav",
        "boundary-native-stdout",
        "boundary-native-stderr",
        "boundary-frozen-case",
        "boundary-metrics",
        "boundary-audit",
        "boundary-provenance"
      ],
      "conditions": [
        {
          "condition": "Independent pre-inference and post-inference validation confirms 16 kHz, mono, signed 16-bit PCM, exactly 160,000 zero-valued samples/320,000 PCM bytes and unchanged silence-input.wav SHA256. It is a distinct input from the primary speech case.",
          "verdict": "passed",
          "observed": "Independent actual WAV parsing confirms 16 kHz mono signed 16-bit PCM, 160000 zero-valued samples and 320000 PCM bytes; input hash unchanged and distinct from primary.",
          "evidenceIds": [
            "boundary-input-wav",
            "boundary-metrics",
            "boundary-audit"
          ]
        },
        {
          "condition": "The declared native bundled-server transcribe process returns exit code 0 within 120 seconds with the same pre-frozen executable/app/model SHA256s and exact arguments, using silence-input.wav rather than the primary audio.",
          "verdict": "passed",
          "observed": "Actual same native model/executable call returned exit=0 in 15876 ms within 120 seconds, with the exact silence input and original arguments.",
          "evidenceIds": [
            "boundary-provenance",
            "boundary-native-stderr"
          ]
        },
        {
          "condition": "Unedited native stdout is empty after whitespace trimming and contains zero lexical transcript tokens; any speech, bracketed music label, speaker identity, explanatory prose or shop detail fails this condition. Native diagnostic stderr is retained separately and cannot be moved into or out of the transcript to improve the outcome.",
          "verdict": "failed",
          "observed": "Unedited native stdout is \" [BLANK_AUDIO]\\n\"; trimmed text is not empty, lexical tokens=1. It is a native blank-audio marker, and the unchanged strict whitespace-only condition fails.",
          "evidenceIds": [
            "boundary-native-stdout",
            "boundary-metrics"
          ]
        },
        {
          "condition": "The separate boundary manifest and native process/output receipts retain the actual input path, original stdout/stderr bytes and their SHA256s, exit code, wall time and case id. No primary transcript, fabricated empty file, repaired text, alternate model or added VAD result substitutes for this native silence call.",
          "verdict": "passed",
          "observed": "Separate boundary working directory, actual native process receipt, distinct input path, raw stdout/stderr hashes and original case id retained; no primary output, fabricated empty file, VAD, retry or repaired result substituted.",
          "evidenceIds": [
            "boundary-provenance",
            "boundary-audit",
            "boundary-native-stdout",
            "boundary-native-stderr"
          ]
        }
      ]
    }
  ],
  "readonlyFiles": [
    {
      "path": "D:\\uAgentKit-lab\\vibe-20261003-native\\fixtures\\speech-input.wav",
      "beforeSha256": "59a144ef862a28687da4fe06801dd8f3a4a4c71202dd61c59588d88419c5fd73",
      "afterSha256": "59a144ef862a28687da4fe06801dd8f3a4a4c71202dd61c59588d88419c5fd73"
    },
    {
      "path": "D:\\uAgentKit-lab\\vibe-20261003-native\\fixtures\\silence-input.wav",
      "beforeSha256": "ee7bea4232762775f8fce9b3e27e4d3948c8ac6a45a87ca769f703d6eed0b448",
      "afterSha256": "ee7bea4232762775f8fce9b3e27e4d3948c8ac6a45a87ca769f703d6eed0b448"
    },
    {
      "path": "D:\\uAgentKit-lab\\vibe-20261003-native\\models\\ggml-tiny.bin",
      "beforeSha256": "be07e048e1e599ad46341c8d2a135645097a538221678b7acdd1b1919c6e1b21",
      "afterSha256": "be07e048e1e599ad46341c8d2a135645097a538221678b7acdd1b1919c6e1b21"
    },
    {
      "path": "D:\\uAgentKit-lab\\vibe-20261003-native\\fixtures\\spoken-reference.txt",
      "beforeSha256": "35123c38d1cea52ccddc932a88f2df58e07ad4882130345a132dcf69c58c63a0",
      "afterSha256": "35123c38d1cea52ccddc932a88f2df58e07ad4882130345a132dcf69c58c63a0"
    }
  ],
  "audit": {
    "method": "Per-case exact input/reference/model/native-executable pre/post SHA256 comparison and actual native subprocess stdout/stderr/version/command receipts; no shared registry or catalog mutation by this task.",
    "readAttempts": [],
    "blockedActions": [],
    "stagedFilesBefore": [],
    "stagedFilesAfter": [],
    "limitations": [
      "Native argument/route provenance is recorded; no complete OS filesystem/network audit, denied-read sandbox or Git staging inspection was performed. Empty attempt/staging arrays are not proof of no other host activity.",
      "Actual non-verbose CLI preserves original parameters; it exposes no inner selected-provider object, ASR decoder tokens or pure inference latency."
    ]
  },
  "usage": {
    "inputTokens": null,
    "outputTokens": null,
    "source": "Native local Whisper ASR speech-transcription inference through the official bundled Vibe server. No chat-LLM API or optional Claude/Ollama analysis calls were made. ASR decoder token counts were not measured."
  },
  "cost": {
    "amount": null,
    "currency": null,
    "scope": "No paid model API or optional remote analysis was used. Local hardware, processing time, download traffic, software/dependency/model/media conditions and total operating cost were not monetarily measured."
  }
}
