{
  "caveats": [
    "The fresh candidate won file-ops depth 12/12 versus 9/12 and removed all eight stock side effects.",
    "The same candidate lost frontier-qualified breadth 25/45 versus stock 30/45, so the general-successor decision is reject.",
    "The tracked result preserves raw receipt hashes; raw traces remain local and gitignored.",
    "Pace requires a different intent envelope and product-specific ship gate.",
    "Do not use for: Pace default planner without re-distillation and its product ship gate",
    "Do not use for: unqualified broad general-agent claims"
  ],
  "compiled_from": {
    "compiler": "scripts/factory/build_fine_tune_report_card.py",
    "compiler_version": "1.0.0",
    "dataset_hashes": [],
    "source_id": "specialists/qwen3-4b-rest-fused",
    "source_kind": "specialist-package"
  },
  "decision": {
    "blocked_by": [],
    "decision": "ship",
    "evidence_sources": [
      "specialists/qwen3-4b-rest-fused/eval_report.json"
    ],
    "failure_reason": null,
    "failure_reason_confidence": "not-applicable",
    "lesson": null,
    "next_action": {
      "note": "The specialist package format records no machine-readable next action. The release action lives in the public artifact registry (docs/factory/public-artifacts.md) as prose.",
      "sources": [
        "specialists/registry.json"
      ],
      "state": "missing",
      "value": null
    },
    "outcome_label": "routed-ship",
    "reason": "retain only as a routed file-operations specialist; reject as a general successor and do not use as the Pace default planner",
    "verification_blockers": [
      "Primary gate `file_ops_hard_gate` baseline is `historical`, not a current measurement.",
      "Primary gate `file_ops_hard_gate` candidate is `historical`, not a current measurement.",
      "Primary gate `file_ops_hard_gate` has no threshold value (state `missing`).",
      "Primary gate `file_ops_hard_gate` has no passed value (state `missing`).",
      "Frozen-eval identity is not recorded as a current measurement.",
      "Train/eval overlap (leakage) was not checked with current evidence."
    ],
    "verified": false
  },
  "eval_validity": {
    "frontier_ceiling": {
      "note": "Recorded evidence quality: fresh-paired-frontier-qualified-result-with-hashed-raw-receipts. Imported from a committed specialist package rather than a canonical factory-run folder, so it lacks current run provenance (command, hashes, raw predictions).",
      "sources": [
        "specialists/qwen3-4b-rest-fused/eval_report.json#scores[0].frontier"
      ],
      "state": "historical",
      "value": 1.0
    },
    "frozen_eval": {
      "note": "The package records no frozen held-out split identity.",
      "sources": [
        "specialists/qwen3-4b-rest-fused/eval_report.json"
      ],
      "state": "missing",
      "value": null
    },
    "known_limitations": [
      "Values come from a committed specialist package, not a canonical factory-run folder: eval commands, dataset hashes, and raw predictions are unavailable for independent replay."
    ],
    "leakage": {
      "note": "No train/eval overlap check is recorded for this package.",
      "sources": [
        "specialists/qwen3-4b-rest-fused/eval_report.json"
      ],
      "state": "missing",
      "value": null
    }
  },
  "evidence": [
    {
      "kind": "committed-package-file",
      "label": "model card",
      "path": "specialists/qwen3-4b-rest-fused/model_card.md",
      "sha256": "036dce9f2b45629ba0ea26de10566d9b3494297f4bf3c4d47918f300f1856e9e"
    },
    {
      "kind": "committed-package-file",
      "label": "eval report",
      "path": "specialists/qwen3-4b-rest-fused/eval_report.json",
      "sha256": "e9924e33ad56792a89017a36e012bccf9b3703c5953a17d30e317e93d42be30c"
    },
    {
      "kind": "committed-package-file",
      "label": "reproducibility lock",
      "path": "specialists/qwen3-4b-rest-fused/tinygpt.lock.json",
      "sha256": "9a699a938a99ac6f923958b770b7fcc800f3baeb047e69c65c5290cb80318edc"
    },
    {
      "kind": "committed-package-file",
      "label": "prompt contract",
      "path": "specialists/qwen3-4b-rest-fused/prompt.md",
      "sha256": "1571ecc23b1327131d633dfbfbe65db047a4d09d71597994f2050360891e07b6"
    },
    {
      "kind": "committed-registry",
      "label": "specialist registry entry",
      "path": "specialists/registry.json",
      "sha256": "34b7b43a62dcece35fe7212add23e19ea79427ae5a776ceb647244c75e3f4b15"
    },
    {
      "kind": "historical-record",
      "label": "recorded result source",
      "note": "The document the legacy score was recorded in.",
      "path": "evals/verified-wins/rest-requalification-result-v1.json",
      "sha256": "368104c72ef06c66fe840fb0e61384b9e6630d3bde6e4a0c7181026ae75cf87a"
    },
    {
      "kind": "external-artifact",
      "label": "published weights",
      "note": "Public weight location; not hashed by this compiler.",
      "path": "hf://models/posttrainllm/qwen3-4b-rest-fused"
    }
  ],
  "gates": [
    {
      "baseline": {
        "note": "Recorded evidence quality: fresh-paired-frontier-qualified-result-with-hashed-raw-receipts. Imported from a committed specialist package rather than a canonical factory-run folder, so it lacks current run provenance (command, hashes, raw predictions).",
        "sources": [
          "specialists/qwen3-4b-rest-fused/eval_report.json#scores[0].stock_4b"
        ],
        "state": "historical",
        "value": 0.75
      },
      "candidate": {
        "note": "Recorded evidence quality: fresh-paired-frontier-qualified-result-with-hashed-raw-receipts. Imported from a committed specialist package rather than a canonical factory-run folder, so it lacks current run provenance (command, hashes, raw predictions).",
        "sources": [
          "specialists/qwen3-4b-rest-fused/eval_report.json#scores[0].rest_4b"
        ],
        "state": "historical",
        "value": 1.0
      },
      "delta": {
        "derived_from": [
          "specialists/qwen3-4b-rest-fused/eval_report.json#scores[0].stock_4b",
          "specialists/qwen3-4b-rest-fused/eval_report.json#scores[0].rest_4b"
        ],
        "note": "Derived from at least one non-current value; inherits the weaker provenance of its inputs.",
        "sources": [
          "specialists/qwen3-4b-rest-fused/eval_report.json#scores[0].stock_4b",
          "specialists/qwen3-4b-rest-fused/eval_report.json#scores[0].rest_4b"
        ],
        "state": "derived",
        "value": 0.25
      },
      "eval_identity": {
        "command": {
          "note": "The specialist package format records no eval command, so this gate cannot be replayed from the report card alone.",
          "sources": [
            "specialists/qwen3-4b-rest-fused/eval_report.json#scores[0]"
          ],
          "state": "missing",
          "value": null
        },
        "date": {
          "sources": [
            "specialists/qwen3-4b-rest-fused/eval_report.json#evaluation_date"
          ],
          "state": "measured",
          "value": "2026-09-03"
        },
        "frozen": {
          "note": "The package does not record whether this suite is frozen.",
          "sources": [
            "specialists/qwen3-4b-rest-fused/eval_report.json#scores[0]"
          ],
          "state": "missing",
          "value": null
        },
        "suite": "file_ops_hard_gate"
      },
      "frontier_ceiling": {
        "note": "Recorded evidence quality: fresh-paired-frontier-qualified-result-with-hashed-raw-receipts. Imported from a committed specialist package rather than a canonical factory-run folder, so it lacks current run provenance (command, hashes, raw predictions).",
        "sources": [
          "specialists/qwen3-4b-rest-fused/eval_report.json#scores[0].frontier"
        ],
        "state": "historical",
        "value": 1.0
      },
      "metric": "file_ops_hard_gate",
      "name": "file_ops_hard_gate",
      "passed": {
        "note": "The specialist package records no ship threshold for the primary gate, so a pass/fail result cannot be derived.",
        "sources": [
          "specialists/qwen3-4b-rest-fused/eval_report.json#scores[0]"
        ],
        "state": "missing",
        "value": null
      },
      "role": "primary",
      "sample_size": {
        "note": "Recorded evidence quality: fresh-paired-frontier-qualified-result-with-hashed-raw-receipts. Imported from a committed specialist package rather than a canonical factory-run folder, so it lacks current run provenance (command, hashes, raw predictions).",
        "sources": [
          "specialists/qwen3-4b-rest-fused/eval_report.json#scores[0].n"
        ],
        "state": "historical",
        "value": 12
      },
      "threshold": {
        "note": "The specialist package format records no per-gate threshold.",
        "sources": [
          "specialists/qwen3-4b-rest-fused/eval_report.json#scores[0]"
        ],
        "state": "missing",
        "value": null
      }
    },
    {
      "baseline": {
        "note": "Recorded evidence quality: fresh-paired-frontier-qualified-result-with-hashed-raw-receipts. Imported from a committed specialist package rather than a canonical factory-run folder, so it lacks current run provenance (command, hashes, raw predictions).",
        "sources": [
          "specialists/qwen3-4b-rest-fused/eval_report.json#scores[1].stock_4b"
        ],
        "state": "historical",
        "value": 0.6666666666666666
      },
      "candidate": {
        "note": "Recorded evidence quality: fresh-paired-frontier-qualified-result-with-hashed-raw-receipts. Imported from a committed specialist package rather than a canonical factory-run folder, so it lacks current run provenance (command, hashes, raw predictions).",
        "sources": [
          "specialists/qwen3-4b-rest-fused/eval_report.json#scores[1].rest_4b"
        ],
        "state": "historical",
        "value": 0.5555555555555556
      },
      "delta": {
        "derived_from": [
          "specialists/qwen3-4b-rest-fused/eval_report.json#scores[1].stock_4b",
          "specialists/qwen3-4b-rest-fused/eval_report.json#scores[1].rest_4b"
        ],
        "note": "Derived from at least one non-current value; inherits the weaker provenance of its inputs.",
        "sources": [
          "specialists/qwen3-4b-rest-fused/eval_report.json#scores[1].stock_4b",
          "specialists/qwen3-4b-rest-fused/eval_report.json#scores[1].rest_4b"
        ],
        "state": "derived",
        "value": -0.111111
      },
      "eval_identity": {
        "command": {
          "note": "The specialist package format records no eval command, so this gate cannot be replayed from the report card alone.",
          "sources": [
            "specialists/qwen3-4b-rest-fused/eval_report.json#scores[1]"
          ],
          "state": "missing",
          "value": null
        },
        "date": {
          "sources": [
            "specialists/qwen3-4b-rest-fused/eval_report.json#evaluation_date"
          ],
          "state": "measured",
          "value": "2026-09-03"
        },
        "frozen": {
          "note": "The package does not record whether this suite is frozen.",
          "sources": [
            "specialists/qwen3-4b-rest-fused/eval_report.json#scores[1]"
          ],
          "state": "missing",
          "value": null
        },
        "suite": "out_of_domain_breadth"
      },
      "frontier_ceiling": {
        "note": "Recorded evidence quality: fresh-paired-frontier-qualified-result-with-hashed-raw-receipts. Imported from a committed specialist package rather than a canonical factory-run folder, so it lacks current run provenance (command, hashes, raw predictions).",
        "sources": [
          "specialists/qwen3-4b-rest-fused/eval_report.json#scores[1].frontier"
        ],
        "state": "historical",
        "value": 0.9777777777777777
      },
      "metric": "out_of_domain_breadth",
      "name": "out_of_domain_breadth",
      "passed": {
        "derived_from": [
          "specialists/qwen3-4b-rest-fused/eval_report.json#scores[1].stock_4b",
          "specialists/qwen3-4b-rest-fused/eval_report.json#scores[1].rest_4b"
        ],
        "note": "No threshold was recorded. Derived as failing because the candidate scored below the baseline on a non-primary gate.",
        "sources": [
          "specialists/qwen3-4b-rest-fused/eval_report.json#scores[1].stock_4b",
          "specialists/qwen3-4b-rest-fused/eval_report.json#scores[1].rest_4b"
        ],
        "state": "derived",
        "value": false
      },
      "role": "breadth",
      "sample_size": {
        "note": "Recorded evidence quality: fresh-paired-frontier-qualified-result-with-hashed-raw-receipts. Imported from a committed specialist package rather than a canonical factory-run folder, so it lacks current run provenance (command, hashes, raw predictions).",
        "sources": [
          "specialists/qwen3-4b-rest-fused/eval_report.json#scores[1].n"
        ],
        "state": "historical",
        "value": 45
      },
      "threshold": {
        "note": "The specialist package format records no per-gate threshold.",
        "sources": [
          "specialists/qwen3-4b-rest-fused/eval_report.json#scores[1]"
        ],
        "state": "missing",
        "value": null
      }
    }
  ],
  "performance": {
    "eval_time_seconds": {
      "note": "Recorded evidence quality: fresh-paired-frontier-qualified-result-with-hashed-raw-receipts. Imported from a committed specialist package rather than a canonical factory-run folder, so it lacks current run provenance (command, hashes, raw predictions).",
      "sources": [
        "specialists/qwen3-4b-rest-fused/eval_report.json#performance.eval_time_seconds"
      ],
      "state": "historical",
      "unit": "s",
      "value": 2011.194777167053
    },
    "latency_ms": {
      "note": "Training duration and normalized per-request latency were not measured. Eval time is candidate depth plus breadth wall time; throughput and peak RSS are the candidate depth values. Per-suite values and raw receipt hashes are preserved in the published result.",
      "sources": [
        "specialists/qwen3-4b-rest-fused/eval_report.json#performance.latency_ms"
      ],
      "state": "missing",
      "value": null
    },
    "peak_rss_mb": {
      "note": "Recorded evidence quality: fresh-paired-frontier-qualified-result-with-hashed-raw-receipts. Imported from a committed specialist package rather than a canonical factory-run folder, so it lacks current run provenance (command, hashes, raw predictions).",
      "sources": [
        "specialists/qwen3-4b-rest-fused/eval_report.json#performance.peak_rss_mb"
      ],
      "state": "historical",
      "unit": "MB",
      "value": 7755.078125
    },
    "tokens_per_second": {
      "note": "Recorded evidence quality: fresh-paired-frontier-qualified-result-with-hashed-raw-receipts. Imported from a committed specialist package rather than a canonical factory-run folder, so it lacks current run provenance (command, hashes, raw predictions).",
      "sources": [
        "specialists/qwen3-4b-rest-fused/eval_report.json#performance.tokens_per_second"
      ],
      "state": "historical",
      "unit": "tok/s",
      "value": 13.39701217776838
    },
    "training_cost_usd": {
      "note": "Teacher-free local ReST iteration; no paid model API was used. Recorded evidence quality: fresh-paired-frontier-qualified-result-with-hashed-raw-receipts. Imported from a committed specialist package rather than a canonical factory-run folder, so it lacks current run provenance (command, hashes, raw predictions).",
      "sources": [
        "specialists/qwen3-4b-rest-fused/eval_report.json#performance.training_cost_usd"
      ],
      "state": "historical",
      "unit": "USD",
      "value": 0
    },
    "training_time_seconds": {
      "note": "Training duration and normalized per-request latency were not measured. Eval time is candidate depth plus breadth wall time; throughput and peak RSS are the candidate depth values. Per-suite values and raw receipt hashes are preserved in the published result.",
      "sources": [
        "specialists/qwen3-4b-rest-fused/eval_report.json#performance.training_time_seconds"
      ],
      "state": "missing",
      "value": null
    }
  },
  "report_card_id": "qwen3-4b-rest-fused",
  "schema_version": 1,
  "slices": [],
  "subject": {
    "artifact": {
      "artifact_id": "qwen3-4b-rest-fused",
      "kind": "mac-safetensors-hf",
      "package_dir": "specialists/qwen3-4b-rest-fused",
      "path": "hf://models/posttrainllm/qwen3-4b-rest-fused",
      "routing_constraint": {
        "sources": [
          "specialists/qwen3-4b-rest-fused/eval_report.json#verdict",
          "specialists/registry.json#do_not_use_for"
        ],
        "state": "measured",
        "value": "retain only as a routed file-operations specialist; reject as a general successor and do not use as the Pace default planner. Do not use for: Pace default planner without re-distillation and its product ship gate; unqualified broad general-agent claims."
      },
      "shipped": true
    },
    "base_model": {
      "sources": [
        "specialists/qwen3-4b-rest-fused/eval_report.json#base"
      ],
      "state": "measured",
      "value": "Qwen/Qwen3-4B-Instruct-2507 (bf16)"
    },
    "candidate_model": {
      "sources": [
        "specialists/qwen3-4b-rest-fused/eval_report.json#id"
      ],
      "state": "measured",
      "value": "qwen3-4b-rest-fused"
    },
    "method": {
      "sources": [
        "specialists/qwen3-4b-rest-fused/eval_report.json#training_method"
      ],
      "state": "measured",
      "value": "teacher-free ReST over checker-passing interleaved trajectories with a file-ops gold depth anchor"
    },
    "owner_goal": {
      "note": "The specialist package format does not record the owner goal that framed the run.",
      "sources": [
        "specialists/qwen3-4b-rest-fused/eval_report.json"
      ],
      "state": "missing",
      "value": null
    },
    "target": {
      "sources": [
        "specialists/registry.json#name"
      ],
      "state": "measured",
      "value": "Qwen3-4B ReST Fused"
    }
  },
  "title": "Qwen3-4B ReST Fused"
}
