{
  "schema": "explainer-film-receipt/1",
  "slug": "visible-reasoning",
  "title": "What a model's written reasoning can and cannot show",
  "page": "explainers.html#visible-reasoning",
  "script": {
    "path": "media/explainers/visible-reasoning/film.json",
    "sha256": "0d91586499fa582c01582fab040b1100f168467d46eeb2c56a669674cbebfcc6"
  },
  "evidence": {
    "path": "media/explainers/visible-reasoning/evidence.json",
    "sha256": "5d833c8950e324283bbb9938924bc70688e7aa1f4f617933e8339fc7b78e152d"
  },
  "render_code": {
    "tools/explainer/film/__init__.py": "1c68de3eaf60d4bd347cf52e0b259326eb4fabeba8c7ec291fc3d16c419a2866",
    "tools/explainer/film/timeline.py": "107a318162298abe409e906a7f6ff3ee56ab984eba120f7678710538284f481b",
    "tools/explainer/film/figures.py": "d80dd3d3975fe6806512608541a5ad76617da3b03a947c6f6565685cfe05f330",
    "tools/explainer/film/plate.py": "43a68d3e2ea36d2337fed0298dec85a318d447e185b8d2c5f99b2089c196cddd",
    "tools/explainer/film/render.py": "de5eb4a123f97decbb1cdcdb6556d931fed7b32c36de8fbca1c026e79a86f7a4",
    "tools/explainer/film/narrate.py": "95a878f114cd5695e55439d07d172c0e6bc0780f40f6c82e8a31cd96193f0c8d"
  },
  "toolchain": {
    "python": "3.12.10",
    "ffmpeg": "ffmpeg version 7.1-essentials_build-www.gyan.dev Copyright (c) 2000-2024 the FFmpeg developers",
    "os": "Windows-11-10.0.26220-SP0",
    "plate": "GLSL 330 on moderngl, headless, GPU"
  },
  "narration": {
    "wav_sha256": "e8b9bbe0f2c746f60903215645b59f833dce49e0ba64590f5b1b0f78a934a111",
    "seconds": 107.84,
    "backend": "qwen3-tts-local",
    "hosted": false,
    "model": "Qwen3-TTS-12Hz-1.7B-Base + the author's fine-tune",
    "fine_tune_sha256": "629342643caee8b5b2df4117f482ee758a447709d5ec6fb79eb6b09140686838",
    "package": "qwen-tts 0.1.1",
    "settings": {
      "dtype": "bfloat16",
      "attn_implementation": "sdpa",
      "do_sample": true,
      "temperature": 0.9,
      "top_k": 50,
      "top_p": 1.0,
      "repetition_penalty": 1.05,
      "subtalker_temperature": 0.9,
      "subtalker_top_k": 50,
      "language": "Auto",
      "speaker": "author",
      "mode": "custom_voice",
      "device": "cuda"
    },
    "label": "A synthesized version of the author's voice, from a model fine-tuned on his recordings with his approval.",
    "asr_check": {
      "engine": "faster-whisper",
      "model": "medium.en",
      "device": "cpu int8",
      "accept_ratio": 0.92,
      "tries": 5,
      "sentences": 22,
      "flagged": 0,
      "mean_ratio": 0.9951
    },
    "loudness": {
      "rule": "explainer-speech-normalise/1",
      "meter": "superstack-bs1770/1",
      "target_lufs": -16.0,
      "before_lufs": -22.03,
      "before_peak_dbfs": -6.03,
      "gain_db": 6.09,
      "limiter": {
        "ceiling_dbfs": -2.0,
        "lookahead_s": 0.02,
        "oversample": 4
      },
      "after_lufs": -16.0,
      "after_peak_dbfs": -2.0,
      "sample_peak_headroom_db": 0.5
    },
    "rate": 48000,
    "reproducible": false,
    "does_not_prove": [
      "An ASR match shows the words were spoken as written; it does not show they sound natural.",
      "Seeds are recorded, but GPU sampling was not checked for bit-exact reruns, so a rerun may differ."
    ]
  },
  "frames": 3280,
  "fps": 30,
  "seconds": 109.33333333333333,
  "frame_chain_sha256": "6b66e4869896bce0b3de64eaf872add93291b19062ba7ca5341f385cea84c33a",
  "outputs": {
    "visible-reasoning.mp4": "28e52bee2f99ca692566161bbce3b7451b291ff08939949a456b10f8d1874181",
    "visible-reasoning.vtt": "fae4d0d2b40a677905460432560598d44d1482dfe7febf233ea5fcc025e4c186",
    "visible-reasoning.srt": "42684f31884751b5e9a05a61b167dd6144ae317b06cad2a81107fab5df6eca49",
    "poster.jpg": "a1e7fd05593cf646102305b30cf2e0880ae4d1abb478856665599818406b6b53",
    "timing.json": "fea617149253a3a465c02f85c4b4be60d63af5008b8efb9731f563762bd32337"
  },
  "does_not_prove": "Matching hashes show these files are the ones this render made from this script, code and narration. They do not show that the explanation is correct or that it teaches. The narration is a synthesized version of the author's voice, sampled from a speech model, so a rerun need not give the same audio bytes."
}
