{
  "source_commit": "fb3ad4e",
  "date": "2026-09-26",
  "scope": "Frozen direct-mode native evaluator before optional thinking mode implementation",
  "libraries_original_path": "lib",
  "sha256": {
    "l2s1": "9cc563af81b61b251c7b7cf3d9ad07a045fd5e5ec0ca60d16168ee1030432fed",
    "evaluate_jsonl": "7fb6e5237188de8161a3025f6b0ce7e16e7b2e3274520a19879e95a115d221ba"
  },
  "hardware": {
    "gpu": "NVIDIA GeForce RTX 3080",
    "vram_mib": 10240,
    "driver": "596.21",
    "cpu": "Intel Core i9-9900K",
    "os": "Linux under WSL2",
    "threads": 4
  },
  "prepared": {
    "suite": "typed",
    "sources": {
      "suite": "typed",
      "files": [
        {
          "repo": "LocalLLaMA/typed-decisions",
          "revision": "c76749ec58bd8c3d2ea706b31c333a9059c38f90",
          "path": "all/test-00000-of-00001.parquet",
          "local": "test-00000-of-00001.parquet",
          "url": "https://huggingface.co/datasets/LocalLLaMA/typed-decisions/resolve/c76749ec58bd8c3d2ea706b31c333a9059c38f90/all/test-00000-of-00001.parquet",
          "sha256": "4f294f218ea1da27f3efef936359389c62ea4d3973a41457732990f1d31b647c"
        },
        {
          "repo": "Luni/laya-jev-benchmark",
          "revision": "d75081b2a4b2ad772793d6a7f5f5b4fdca00d557",
          "path": "bench/eval.py",
          "local": "eval.py",
          "url": "https://huggingface.co/datasets/Luni/laya-jev-benchmark/resolve/d75081b2a4b2ad772793d6a7f5f5b4fdca00d557/bench/eval.py",
          "sha256": "da874d9bdfd3ae1793c6a5d9e7def69349afc19e73874a2a0200c8b8eb90346c"
        }
      ]
    },
    "available_cases": 400,
    "logical_cases": 400,
    "repeats": 1,
    "measured_cases": 400,
    "decisions": 2000,
    "limited": false,
    "files": {
      "requests.jsonl": "a677142b77f72f4445d79d10213d79e7de564f223553bb3a4c88105a181f7e6d",
      "cases-with-gold.jsonl": "2bd01c323fd2c6e7c767f8db4119f8121442cc62b9774ddd113702a5607cf906"
    },
    "adapter_sha256": "44f59fb153e2767e18cf30ad76b7817c1fa83373618588844c3abd9626358d74",
    "ordering": "source order, immediate repeats per case",
    "protocol": "Only id/state/instruction/criteria enter inference. No gold, rationale or workflow metadata. Repeated identical cases measure preparation caching, not a persistent KV or answer cache. Probe grounding outputs also supply overconfidence checks; no second inference for the same check."
  },
  "postprocess_scorer_sha256": "f62991bf827344e9fcd23a43828668ee5b1a994d18c9f5dc996cd06d91ac1621",
  "runs": {
    "gemma4-e2b": {
      "adapter_sha256": "07d18e1a7590a006df9fb957ea8dd9e1bd434b7e1fcd5f648ea6eae06724c5fd",
      "command": [
        "evaluate_jsonl",
        "--model",
        "gemma-4-E2B-it-Q8_0.gguf",
        "--input",
        "requests.jsonl",
        "--output",
        "results/typed-decisions-20260926/gemma4-e2b/predictions.jsonl",
        "--context",
        "8192",
        "--batch",
        "256",
        "--threads",
        "4",
        "--execution-mode",
        "fresh",
        "--prompt-layout",
        "legacy",
        "--parallel-width",
        "4",
        "--request-batch-size",
        "1",
        "--model-load-mode",
        "read",
        "--preparation-cache-bytes",
        "0",
        "--preparation-cache-entries",
        "128",
        "--warmup",
        "--cuda"
      ],
      "evaluator_sha256": "7fb6e5237188de8161a3025f6b0ce7e16e7b2e3274520a19879e95a115d221ba",
      "model_sha256": "996d08777aadc6bfd3c7375ef70ba25a0f55240075860754fdb18d6d860aa63a",
      "prepared_manifest_sha256": "729821e37249791f474825903bbcfe6e7048c61dc69e06054dcabca478ee509b",
      "timing": "One resident model, untimed warmup then preparation-cache clear; load excluded. No answer cache."
    },
    "qwen3-06b": {
      "adapter_sha256": "07d18e1a7590a006df9fb957ea8dd9e1bd434b7e1fcd5f648ea6eae06724c5fd",
      "command": [
        "evaluate_jsonl",
        "--model",
        "Qwen3-0.6B-Q8_0.gguf",
        "--input",
        "requests.jsonl",
        "--output",
        "results/typed-decisions-20260926/qwen3-06b/predictions.jsonl",
        "--context",
        "8192",
        "--batch",
        "256",
        "--threads",
        "4",
        "--execution-mode",
        "fresh",
        "--prompt-layout",
        "legacy",
        "--parallel-width",
        "4",
        "--request-batch-size",
        "1",
        "--model-load-mode",
        "read",
        "--preparation-cache-bytes",
        "0",
        "--preparation-cache-entries",
        "128",
        "--warmup",
        "--cuda"
      ],
      "evaluator_sha256": "7fb6e5237188de8161a3025f6b0ce7e16e7b2e3274520a19879e95a115d221ba",
      "model_sha256": "9465e63a22add5354d9bb4b99e90117043c7124007664907259bd16d043bb031",
      "prepared_manifest_sha256": "729821e37249791f474825903bbcfe6e7048c61dc69e06054dcabca478ee509b",
      "timing": "One resident model, untimed warmup then preparation-cache clear; load excluded. No answer cache."
    }
  },
  "public_export": {
    "path_normalization": "Absolute native-library and command paths are reduced to basenames. Runs are added from checked-in model run metadata. Measurements, revisions, command flags and hashes are unchanged."
  }
}
