model-conversion : add --add-bos to run org model script (#29558)

This commit adds an optional --add-bos token command line option to the
run-org-model.py script.

The motivation for this is that there are models, for example Gemma4,
that explicitely set the add_bos value to true in llama-vocab.cpp even
if the original model does not set this value to True.

It would be nice to be able to force the models to agree on the bos
token so that logit verification can proceed.

Refs: https://github.com/ggml-org/llama.cpp/pull/21500
This commit is contained in:
Daniel Bevenius
2026-09-30 12:52:02 +02:00
committed by GitHub
parent 4a096b8ff6
commit 8df332de1b
@@ -19,6 +19,8 @@ def parse_arguments():
parser.add_argument("--prompt-file", "-f", help="Optional prompt file", required=False)
parser.add_argument("--verbose", "-v", action="store_true", help="Enable verbose debug output")
parser.add_argument("--device", "-d", help="Device to use (cpu, cuda, mps, auto)", default="auto")
parser.add_argument("--add-bos", action=argparse.BooleanOptionalAction, default=None,
help="Override BOS token setting (default: use model's own setting)")
return parser.parse_args()
def load_model_and_tokenizer(model_path, device="auto"):
@@ -119,6 +121,9 @@ def main():
model, tokenizer, config = load_model_and_tokenizer(model_path, args.device)
if args.add_bos is not None and hasattr(tokenizer, "add_bos_token"):
tokenizer.add_bos_token = args.add_bos
if args.verbose:
enable_torch_debugging(model)