mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-03 03:17:32 -05:00
cont : cleanup
This commit is contained in:
@@ -221,7 +221,6 @@ jobs:
|
||||
# 7z x "-o${env:RUNNER_TEMP}" $env:RUNNER_TEMP/sde.tar
|
||||
# $sde = $(join-path $env:RUNNER_TEMP sde-external-${env:SDE_VERSION}-win/sde.exe)
|
||||
# cd build
|
||||
# $env:LLAMA_SKIP_TESTS_SLOW_ON_EMULATOR = 1
|
||||
# & $sde -future -- ctest -L main -C Release --verbose --timeout 900
|
||||
|
||||
- name: ccache-clear
|
||||
|
||||
@@ -312,7 +312,7 @@ common_peg_parser analyze_tools::build_tool_parser_tag_json(parser_build_context
|
||||
foreach_function(inputs.tools, [&](const json & tool) {
|
||||
const auto & func = tool.at("function");
|
||||
std::string name = func.at("name");
|
||||
const auto & schema = func.contains("parameters") ? func.at("parameters") : json::object();
|
||||
auto schema = common_chat_tool_parameters(func);
|
||||
|
||||
// Build call_id parser based on position (if supported)
|
||||
bool have_call_id = false;
|
||||
|
||||
@@ -640,7 +640,7 @@ common_peg_parser common_chat_peg_builder::build_json_tools_function_is_key(
|
||||
}
|
||||
const auto & function = tool_def.at("function");
|
||||
std::string name = function.at("name");
|
||||
ordered_json params = function.contains("parameters") ? function.at("parameters") : ordered_json::object();
|
||||
ordered_json params = common_chat_tool_parameters(function);
|
||||
|
||||
// Build inner object fields
|
||||
std::vector<common_peg_parser> inner_fields;
|
||||
@@ -726,7 +726,7 @@ common_peg_parser common_chat_peg_builder::build_json_tools_nested_keys(
|
||||
}
|
||||
const auto & function = tool_def.at("function");
|
||||
std::string name = function.at("name");
|
||||
ordered_json params = function.contains("parameters") ? function.at("parameters") : ordered_json::object();
|
||||
ordered_json params = common_chat_tool_parameters(function);
|
||||
|
||||
auto nested_name = literal("\"" + nested_name_field + "\"") + space() + literal(":") + space() +
|
||||
atomic(literal("\"") + tool_name(literal(name)) + literal("\""));
|
||||
@@ -795,7 +795,7 @@ common_peg_parser common_chat_peg_builder::build_json_tools_flat_keys(
|
||||
}
|
||||
const auto & function = tool_def.at("function");
|
||||
std::string name = function.at("name");
|
||||
ordered_json params = function.contains("parameters") ? function.at("parameters") : ordered_json::object();
|
||||
ordered_json params = common_chat_tool_parameters(function);
|
||||
|
||||
auto tool_name_ = name_key_parser + space() + literal(":") + space() +
|
||||
atomic(literal("\"") + tool_name(literal(name)) + literal("\""));
|
||||
@@ -1230,3 +1230,14 @@ void common_chat_peg_minimax_m3_mapper::visit(const common_peg_ast_arena & arena
|
||||
visit(arena, child_id);
|
||||
}
|
||||
}
|
||||
|
||||
common_json common_chat_tool_parameters(const common_json & function) {
|
||||
if (function.contains("parameters") && function.at("parameters").is_object() && !function.at("parameters").empty()) {
|
||||
return function.at("parameters");
|
||||
}
|
||||
auto schema = common_json::object();
|
||||
schema["type"] = "object";
|
||||
schema["properties"] = common_json::object();
|
||||
schema["additionalProperties"] = false;
|
||||
return schema;
|
||||
}
|
||||
|
||||
@@ -218,3 +218,7 @@ struct tagged_peg_parser {
|
||||
|
||||
tagged_peg_parser build_tagged_peg_parser(
|
||||
const std::function<common_peg_parser(common_peg_parser_builder & builder)> & fn);
|
||||
|
||||
// The parameters schema of a tool for its arguments parser. Like the OpenAI API, a missing or empty
|
||||
// "parameters" means the tool takes no arguments, not that any value is accepted.
|
||||
common_json common_chat_tool_parameters(const common_json & function);
|
||||
|
||||
@@ -347,7 +347,6 @@ private:
|
||||
std::vector<std::string> _errors;
|
||||
std::vector<std::string> _warnings;
|
||||
|
||||
// every schema given to resolve_refs() or add_schema() is parsed into this document, so their $refs resolve through each other
|
||||
common_schema_document _doc;
|
||||
|
||||
template <typename T>
|
||||
@@ -681,19 +680,19 @@ private:
|
||||
return out.str();
|
||||
}
|
||||
|
||||
std::string _resolve_ref(const common_schema_ref & ref) {
|
||||
auto it = ref.ref.find('#');
|
||||
std::string ref_fragment = it != std::string::npos ? ref.ref.substr(it + 1) : ref.ref;
|
||||
std::string _resolve_ref(const common_schema_ref & schema) {
|
||||
auto it = schema.ref.find('#');
|
||||
std::string ref_fragment = it != std::string::npos ? schema.ref.substr(it + 1) : schema.ref;
|
||||
static const std::regex nonalphanumeric_regex(R"([^a-zA-Z0-9-]+)");
|
||||
std::string ref_name = "ref" + std::regex_replace(ref_fragment, nonalphanumeric_regex, "-");
|
||||
if (_rules.find(ref_name) == _rules.end() && _refs_being_resolved.find(ref.ref) == _refs_being_resolved.end()) {
|
||||
if (!ref.target) {
|
||||
_errors.push_back("Unresolved $ref " + ref.ref);
|
||||
if (_rules.find(ref_name) == _rules.end() && _refs_being_resolved.find(schema.ref) == _refs_being_resolved.end()) {
|
||||
if (!schema.target) {
|
||||
_errors.push_back("Unresolved $ref " + schema.ref);
|
||||
return "";
|
||||
}
|
||||
_refs_being_resolved.insert(ref.ref);
|
||||
ref_name = visit(*ref.target, ref_name);
|
||||
_refs_being_resolved.erase(ref.ref);
|
||||
_refs_being_resolved.insert(schema.ref);
|
||||
ref_name = visit(*schema.target, ref_name);
|
||||
_refs_being_resolved.erase(schema.ref);
|
||||
}
|
||||
return ref_name;
|
||||
}
|
||||
@@ -848,7 +847,7 @@ public:
|
||||
return _add_primitive(rule_name == "root" ? "root" : type, PRIMITIVE_RULES.at(type));
|
||||
}
|
||||
|
||||
// An allOf merged the way the Python converter does it: the properties of a direct component are required, those of a nested anyOf are optional, enums are intersected.
|
||||
// An allOf merges its components: the properties of a direct component are required, those of a nested anyOf are optional, enums are intersected.
|
||||
std::string _visit_all_of(const common_schema_all_of & schema, const std::string & name, const std::string & rule_name) {
|
||||
std::unordered_set<std::string> required;
|
||||
std::vector<std::pair<std::string, const common_schema *>> properties;
|
||||
@@ -988,9 +987,6 @@ public:
|
||||
return _visit_primitive(rule_name, "null");
|
||||
case COMMON_SCHEMA_KIND_ANY:
|
||||
return _add_rule(rule_name, _add_primitive("value", PRIMITIVE_RULES.at("value")));
|
||||
case COMMON_SCHEMA_KIND_NONE:
|
||||
_errors.push_back("No value satisfies the schema of " + rule_name);
|
||||
return "";
|
||||
}
|
||||
return "";
|
||||
}
|
||||
|
||||
@@ -1,10 +1,8 @@
|
||||
#include "json-schema.h"
|
||||
#include "common.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
#include <map>
|
||||
#include <set>
|
||||
#include <stdexcept>
|
||||
#include <string>
|
||||
#include <unordered_set>
|
||||
@@ -350,602 +348,6 @@ common_schema_ptr common_schema_parse(const common_json & schema, common_schema_
|
||||
return common_schema_parser(schema, doc).parse();
|
||||
}
|
||||
|
||||
class common_schema_optimizer {
|
||||
common_schema_document & doc_;
|
||||
|
||||
// set when a $ref got replaced by the none its target became, the pass is then repeated
|
||||
bool changed_ = false;
|
||||
|
||||
// the target pairs being intersected further up, a recursive schema would otherwise never bottom out
|
||||
std::set<std::pair<const common_schema *, const common_schema *>> active_;
|
||||
|
||||
template <typename T>
|
||||
static const T & as(const common_schema & node) {
|
||||
return static_cast<const T &>(node);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
static T & as(common_schema & node) {
|
||||
return static_cast<T &>(node);
|
||||
}
|
||||
|
||||
static bool is(const common_schema & node, common_schema_kind kind) {
|
||||
return node.kind() == kind;
|
||||
}
|
||||
|
||||
static bool is(const common_schema_ptr & node, common_schema_kind kind) {
|
||||
return node && node->kind() == kind;
|
||||
}
|
||||
|
||||
static common_schema_ptr none() {
|
||||
return std::make_unique<common_schema_none>();
|
||||
}
|
||||
|
||||
// Follows a chain of $refs to a node of another kind.
|
||||
// Gives nullptr for a target that is being rewritten right now, or a chain that loops back on itself.
|
||||
const common_schema * resolve(const common_schema & node) const {
|
||||
const common_schema * cur = &node;
|
||||
for (size_t hops = 0; is(*cur, COMMON_SCHEMA_KIND_REF); hops++) {
|
||||
if (hops > doc_.refs.size()) {
|
||||
return nullptr;
|
||||
}
|
||||
auto it = doc_.refs.find(as<common_schema_ref>(*cur).ref);
|
||||
if (it == doc_.refs.end() || !it->second) {
|
||||
return nullptr;
|
||||
}
|
||||
cur = it->second.get();
|
||||
}
|
||||
return cur;
|
||||
}
|
||||
|
||||
static std::vector<common_schema_ptr> clone_all(const std::vector<common_schema_ptr> & nodes) {
|
||||
std::vector<common_schema_ptr> out;
|
||||
for (const auto & node : nodes) {
|
||||
out.push_back(clone(*node));
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
static common_schema_ptr clone(const common_schema & node) {
|
||||
switch (node.kind()) {
|
||||
case COMMON_SCHEMA_KIND_ANY: return std::make_unique<common_schema_any>();
|
||||
case COMMON_SCHEMA_KIND_NONE: return none();
|
||||
case COMMON_SCHEMA_KIND_NULL: return std::make_unique<common_schema_null>();
|
||||
case COMMON_SCHEMA_KIND_BOOLEAN: return std::make_unique<common_schema_boolean>();
|
||||
case COMMON_SCHEMA_KIND_NUMBER: return std::make_unique<common_schema_number>();
|
||||
case COMMON_SCHEMA_KIND_INTEGER: return std::make_unique<common_schema_integer>(as<common_schema_integer>(node));
|
||||
case COMMON_SCHEMA_KIND_STRING: return std::make_unique<common_schema_string>(as<common_schema_string>(node));
|
||||
case COMMON_SCHEMA_KIND_CONST: return std::make_unique<common_schema_const>(as<common_schema_const>(node).value);
|
||||
case COMMON_SCHEMA_KIND_REF: {
|
||||
const auto & ref = as<common_schema_ref>(node);
|
||||
auto out = std::make_unique<common_schema_ref>(ref.ref);
|
||||
out->target = ref.target;
|
||||
return out;
|
||||
}
|
||||
case COMMON_SCHEMA_KIND_ENUM: {
|
||||
auto out = std::make_unique<common_schema_enum>();
|
||||
out->values = as<common_schema_enum>(node).values;
|
||||
return out;
|
||||
}
|
||||
case COMMON_SCHEMA_KIND_ANY_OF: {
|
||||
auto out = std::make_unique<common_schema_any_of>();
|
||||
out->children = clone_all(as<common_schema_any_of>(node).children);
|
||||
return out;
|
||||
}
|
||||
case COMMON_SCHEMA_KIND_ALL_OF: {
|
||||
auto out = std::make_unique<common_schema_all_of>();
|
||||
out->children = clone_all(as<common_schema_all_of>(node).children);
|
||||
return out;
|
||||
}
|
||||
case COMMON_SCHEMA_KIND_ARRAY: {
|
||||
const auto & arr = as<common_schema_array>(node);
|
||||
auto out = std::make_unique<common_schema_array>();
|
||||
out->items = clone(*arr.items);
|
||||
out->min_items = arr.min_items;
|
||||
out->max_items = arr.max_items;
|
||||
return out;
|
||||
}
|
||||
case COMMON_SCHEMA_KIND_TUPLE: {
|
||||
auto out = std::make_unique<common_schema_tuple>();
|
||||
out->items = clone_all(as<common_schema_tuple>(node).items);
|
||||
return out;
|
||||
}
|
||||
case COMMON_SCHEMA_KIND_OBJECT: {
|
||||
const auto & obj = as<common_schema_object>(node);
|
||||
auto out = std::make_unique<common_schema_object>();
|
||||
for (const auto & prop : obj.properties) {
|
||||
out->properties.push_back({prop.name, clone(*prop.schema), prop.required});
|
||||
}
|
||||
if (obj.additional_properties) {
|
||||
out->additional_properties = clone(*obj.additional_properties);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
}
|
||||
return none();
|
||||
}
|
||||
|
||||
static bool contains(const std::vector<common_json> & values, const common_json & value) {
|
||||
return std::any_of(values.begin(), values.end(), [&](const common_json & v) { return v == value; });
|
||||
}
|
||||
|
||||
// the values a const or enum accepts
|
||||
static bool get_values(const common_schema & node, std::vector<common_json> & values) {
|
||||
if (is(node, COMMON_SCHEMA_KIND_CONST)) {
|
||||
values.push_back(as<common_schema_const>(node).value);
|
||||
return true;
|
||||
}
|
||||
if (is(node, COMMON_SCHEMA_KIND_ENUM)) {
|
||||
values = as<common_schema_enum>(node).values;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// a const for one value, none for no value at all
|
||||
static common_schema_ptr make_values(std::vector<common_json> values) {
|
||||
if (values.empty()) {
|
||||
return none();
|
||||
}
|
||||
if (values.size() == 1) {
|
||||
return std::make_unique<common_schema_const>(std::move(values[0]));
|
||||
}
|
||||
auto node = std::make_unique<common_schema_enum>();
|
||||
node->values = std::move(values);
|
||||
return node;
|
||||
}
|
||||
|
||||
// The union of rewritten alternatives.
|
||||
static common_schema_ptr make_any_of(std::vector<common_schema_ptr> children) {
|
||||
std::vector<common_schema_ptr> flat;
|
||||
for (auto & child : children) {
|
||||
if (is(child, COMMON_SCHEMA_KIND_ANY_OF)) {
|
||||
// a nested one is flat already
|
||||
for (auto & c : as<common_schema_any_of>(*child).children) {
|
||||
flat.push_back(std::move(c));
|
||||
}
|
||||
} else {
|
||||
flat.push_back(std::move(child));
|
||||
}
|
||||
}
|
||||
std::vector<common_schema_ptr> out;
|
||||
for (auto & child : flat) {
|
||||
if (is(child, COMMON_SCHEMA_KIND_ANY)) {
|
||||
return std::make_unique<common_schema_any>();
|
||||
}
|
||||
if (!is(child, COMMON_SCHEMA_KIND_NONE)) {
|
||||
out.push_back(std::move(child));
|
||||
}
|
||||
}
|
||||
if (out.empty()) {
|
||||
return none();
|
||||
}
|
||||
if (out.size() == 1) {
|
||||
return std::move(out[0]);
|
||||
}
|
||||
auto node = std::make_unique<common_schema_any_of>();
|
||||
node->children = std::move(out);
|
||||
return node;
|
||||
}
|
||||
|
||||
// An array with rewritten items.
|
||||
static common_schema_ptr make_array(common_schema_ptr items, int min_items, int max_items) {
|
||||
if ((max_items >= 0 && min_items > max_items) || (is(items, COMMON_SCHEMA_KIND_NONE) && min_items > 0)) {
|
||||
return none();
|
||||
}
|
||||
if (max_items == 0 || is(items, COMMON_SCHEMA_KIND_NONE)) {
|
||||
// only [] is left
|
||||
return std::make_unique<common_schema_tuple>();
|
||||
}
|
||||
auto node = std::make_unique<common_schema_array>();
|
||||
node->items = std::move(items);
|
||||
node->min_items = min_items;
|
||||
node->max_items = max_items;
|
||||
return node;
|
||||
}
|
||||
|
||||
// An object with rewritten property schemas.
|
||||
static common_schema_ptr make_object(std::vector<common_schema_property> properties, common_schema_ptr additional) {
|
||||
auto node = std::make_unique<common_schema_object>();
|
||||
for (auto & prop : properties) {
|
||||
if (is(prop.schema, COMMON_SCHEMA_KIND_NONE)) {
|
||||
// it can never be present
|
||||
if (prop.required) {
|
||||
return none();
|
||||
}
|
||||
continue;
|
||||
}
|
||||
node->properties.push_back(std::move(prop));
|
||||
}
|
||||
if (additional && !is(additional, COMMON_SCHEMA_KIND_NONE)) {
|
||||
node->additional_properties = std::move(additional);
|
||||
}
|
||||
return node;
|
||||
}
|
||||
|
||||
std::vector<common_schema_ptr> rewrite_all(std::vector<common_schema_ptr> nodes) {
|
||||
for (auto & node : nodes) {
|
||||
node = rewrite(std::move(node));
|
||||
}
|
||||
return nodes;
|
||||
}
|
||||
|
||||
// Rewrites a node bottom-up.
|
||||
common_schema_ptr rewrite(common_schema_ptr node) {
|
||||
switch (node->kind()) {
|
||||
case COMMON_SCHEMA_KIND_ANY:
|
||||
case COMMON_SCHEMA_KIND_NONE:
|
||||
case COMMON_SCHEMA_KIND_CONST:
|
||||
case COMMON_SCHEMA_KIND_NULL:
|
||||
case COMMON_SCHEMA_KIND_BOOLEAN:
|
||||
case COMMON_SCHEMA_KIND_NUMBER:
|
||||
return node;
|
||||
case COMMON_SCHEMA_KIND_REF: {
|
||||
const common_schema * target = resolve(*node);
|
||||
if (target && is(*target, COMMON_SCHEMA_KIND_NONE)) {
|
||||
changed_ = true;
|
||||
return none();
|
||||
}
|
||||
return node;
|
||||
}
|
||||
case COMMON_SCHEMA_KIND_ENUM: {
|
||||
std::vector<common_json> values;
|
||||
for (const auto & v : as<common_schema_enum>(*node).values) {
|
||||
if (!contains(values, v)) {
|
||||
values.push_back(v);
|
||||
}
|
||||
}
|
||||
return make_values(std::move(values));
|
||||
}
|
||||
case COMMON_SCHEMA_KIND_INTEGER: {
|
||||
const auto & i = as<common_schema_integer>(*node);
|
||||
return i.minimum > i.maximum ? none() : std::move(node);
|
||||
}
|
||||
case COMMON_SCHEMA_KIND_STRING: {
|
||||
const auto & s = as<common_schema_string>(*node);
|
||||
return s.max_length >= 0 && s.min_length > s.max_length ? none() : std::move(node);
|
||||
}
|
||||
case COMMON_SCHEMA_KIND_ANY_OF:
|
||||
return make_any_of(rewrite_all(std::move(as<common_schema_any_of>(*node).children)));
|
||||
case COMMON_SCHEMA_KIND_ALL_OF:
|
||||
return intersect_all(rewrite_all(std::move(as<common_schema_all_of>(*node).children)));
|
||||
case COMMON_SCHEMA_KIND_ARRAY: {
|
||||
auto & arr = as<common_schema_array>(*node);
|
||||
return make_array(rewrite(std::move(arr.items)), arr.min_items, arr.max_items);
|
||||
}
|
||||
case COMMON_SCHEMA_KIND_TUPLE: {
|
||||
auto & tup = as<common_schema_tuple>(*node);
|
||||
tup.items = rewrite_all(std::move(tup.items));
|
||||
for (const auto & item : tup.items) {
|
||||
if (is(item, COMMON_SCHEMA_KIND_NONE)) {
|
||||
return none();
|
||||
}
|
||||
}
|
||||
return node;
|
||||
}
|
||||
case COMMON_SCHEMA_KIND_OBJECT: {
|
||||
auto & obj = as<common_schema_object>(*node);
|
||||
for (auto & prop : obj.properties) {
|
||||
prop.schema = rewrite(std::move(prop.schema));
|
||||
}
|
||||
if (obj.additional_properties) {
|
||||
obj.additional_properties = rewrite(std::move(obj.additional_properties));
|
||||
}
|
||||
return make_object(std::move(obj.properties), std::move(obj.additional_properties));
|
||||
}
|
||||
}
|
||||
return node;
|
||||
}
|
||||
|
||||
static common_schema_ptr make_all_of(std::vector<common_schema_ptr> children) {
|
||||
if (children.size() == 1) {
|
||||
return std::move(children[0]);
|
||||
}
|
||||
auto node = std::make_unique<common_schema_all_of>();
|
||||
node->children = std::move(children);
|
||||
return node;
|
||||
}
|
||||
|
||||
// what is left when two nodes cannot be combined
|
||||
static common_schema_ptr irreducible(const common_schema & a, const common_schema & b) {
|
||||
std::vector<common_schema_ptr> children;
|
||||
children.push_back(clone(a));
|
||||
children.push_back(clone(b));
|
||||
return make_all_of(std::move(children));
|
||||
}
|
||||
|
||||
// The intersection of rewritten nodes, an allOf of those that could not be combined.
|
||||
common_schema_ptr intersect_all(std::vector<common_schema_ptr> children) {
|
||||
std::vector<common_schema_ptr> flat;
|
||||
for (auto & child : children) {
|
||||
if (is(child, COMMON_SCHEMA_KIND_ALL_OF)) {
|
||||
for (auto & c : as<common_schema_all_of>(*child).children) {
|
||||
flat.push_back(std::move(c));
|
||||
}
|
||||
} else {
|
||||
flat.push_back(std::move(child));
|
||||
}
|
||||
}
|
||||
|
||||
std::vector<common_schema_ptr> residual;
|
||||
for (auto & child : flat) {
|
||||
bool merged = false;
|
||||
for (auto & r : residual) {
|
||||
auto cand = intersect(*r, *child);
|
||||
if (is(cand, COMMON_SCHEMA_KIND_NONE)) {
|
||||
return none();
|
||||
}
|
||||
if (!is(cand, COMMON_SCHEMA_KIND_ALL_OF)) {
|
||||
r = std::move(cand);
|
||||
merged = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!merged) {
|
||||
residual.push_back(std::move(child));
|
||||
}
|
||||
}
|
||||
if (residual.empty()) {
|
||||
return std::make_unique<common_schema_any>();
|
||||
}
|
||||
return make_all_of(std::move(residual));
|
||||
}
|
||||
|
||||
static const common_schema_property * find_property(const common_schema_object & obj, const std::string & name) {
|
||||
for (const auto & prop : obj.properties) {
|
||||
if (prop.name == name) {
|
||||
return ∝
|
||||
}
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
// Properties are merged the way json_schema_to_grammar() merges an allOf: a property the other
|
||||
// side does not list survives, a side that is closed does not shut it out.
|
||||
common_schema_ptr intersect_object(const common_schema_object & oa, const common_schema_object & ob) {
|
||||
std::vector<common_schema_property> properties;
|
||||
auto add = [&](const common_schema_object & x, const common_schema_object & y, bool skip_shared) {
|
||||
for (const auto & px : x.properties) {
|
||||
const auto * py = find_property(y, px.name);
|
||||
if (py && skip_shared) {
|
||||
continue;
|
||||
}
|
||||
common_schema_property prop;
|
||||
prop.name = px.name;
|
||||
prop.required = px.required || (py && py->required);
|
||||
if (py) {
|
||||
prop.schema = intersect(*px.schema, *py->schema);
|
||||
} else if (y.additional_properties) {
|
||||
prop.schema = intersect(*px.schema, *y.additional_properties);
|
||||
} else {
|
||||
prop.schema = clone(*px.schema);
|
||||
}
|
||||
properties.push_back(std::move(prop));
|
||||
}
|
||||
};
|
||||
add(oa, ob, false);
|
||||
add(ob, oa, true);
|
||||
|
||||
common_schema_ptr additional;
|
||||
if (oa.additional_properties && ob.additional_properties) {
|
||||
additional = intersect(*oa.additional_properties, *ob.additional_properties);
|
||||
}
|
||||
return make_object(std::move(properties), std::move(additional));
|
||||
}
|
||||
|
||||
// The intersection of two rewritten nodes, an allOf of both where they cannot be combined.
|
||||
common_schema_ptr intersect(const common_schema & a, const common_schema & b) {
|
||||
if (is(a, COMMON_SCHEMA_KIND_ANY)) {
|
||||
return clone(b);
|
||||
}
|
||||
if (is(b, COMMON_SCHEMA_KIND_ANY)) {
|
||||
return clone(a);
|
||||
}
|
||||
if (is(a, COMMON_SCHEMA_KIND_NONE) || is(b, COMMON_SCHEMA_KIND_NONE)) {
|
||||
return none();
|
||||
}
|
||||
if (is(a, COMMON_SCHEMA_KIND_REF) || is(b, COMMON_SCHEMA_KIND_REF)) {
|
||||
if (is(a, COMMON_SCHEMA_KIND_REF) && is(b, COMMON_SCHEMA_KIND_REF) && as<common_schema_ref>(a).ref == as<common_schema_ref>(b).ref) {
|
||||
return clone(a);
|
||||
}
|
||||
const common_schema * ta = resolve(a);
|
||||
const common_schema * tb = resolve(b);
|
||||
if (!ta || !tb) {
|
||||
return irreducible(a, b);
|
||||
}
|
||||
auto key = std::make_pair(ta, tb);
|
||||
if (!active_.insert(key).second) {
|
||||
// the same pair is already being intersected further up, a recursive schema
|
||||
return irreducible(a, b);
|
||||
}
|
||||
auto out = intersect(*ta, *tb);
|
||||
active_.erase(key);
|
||||
return out;
|
||||
}
|
||||
if (is(a, COMMON_SCHEMA_KIND_ANY_OF) || is(b, COMMON_SCHEMA_KIND_ANY_OF)) {
|
||||
// (a1 | a2) & b = (a1 & b) | (a2 & b)
|
||||
std::vector<common_schema_ptr> alts;
|
||||
if (is(a, COMMON_SCHEMA_KIND_ANY_OF)) {
|
||||
for (const auto & child : as<common_schema_any_of>(a).children) {
|
||||
alts.push_back(intersect(*child, b));
|
||||
}
|
||||
} else {
|
||||
for (const auto & child : as<common_schema_any_of>(b).children) {
|
||||
alts.push_back(intersect(a, *child));
|
||||
}
|
||||
}
|
||||
return make_any_of(std::move(alts));
|
||||
}
|
||||
if (is(a, COMMON_SCHEMA_KIND_ALL_OF) || is(b, COMMON_SCHEMA_KIND_ALL_OF)) {
|
||||
std::vector<common_schema_ptr> parts;
|
||||
for (const auto * node : {&a, &b}) {
|
||||
if (is(*node, COMMON_SCHEMA_KIND_ALL_OF)) {
|
||||
for (const auto & child : as<common_schema_all_of>(*node).children) {
|
||||
parts.push_back(clone(*child));
|
||||
}
|
||||
} else {
|
||||
parts.push_back(clone(*node));
|
||||
}
|
||||
}
|
||||
return intersect_all(std::move(parts));
|
||||
}
|
||||
|
||||
std::vector<common_json> va;
|
||||
std::vector<common_json> vb;
|
||||
bool values_a = get_values(a, va);
|
||||
bool values_b = get_values(b, vb);
|
||||
if (values_a && values_b) {
|
||||
std::vector<common_json> common;
|
||||
for (const auto & v : va) {
|
||||
if (contains(vb, v)) {
|
||||
common.push_back(v);
|
||||
}
|
||||
}
|
||||
return make_values(std::move(common));
|
||||
}
|
||||
if (values_a || values_b) {
|
||||
// whether a value fits a schema is not checked
|
||||
return irreducible(a, b);
|
||||
}
|
||||
|
||||
if (a.kind() != b.kind()) {
|
||||
if (is(a, COMMON_SCHEMA_KIND_NUMBER) && is(b, COMMON_SCHEMA_KIND_INTEGER)) {
|
||||
return clone(b);
|
||||
}
|
||||
if (is(a, COMMON_SCHEMA_KIND_INTEGER) && is(b, COMMON_SCHEMA_KIND_NUMBER)) {
|
||||
return clone(a);
|
||||
}
|
||||
if ((is(a, COMMON_SCHEMA_KIND_ARRAY) && is(b, COMMON_SCHEMA_KIND_TUPLE)) || (is(a, COMMON_SCHEMA_KIND_TUPLE) && is(b, COMMON_SCHEMA_KIND_ARRAY))) {
|
||||
return irreducible(a, b);
|
||||
}
|
||||
return none();
|
||||
}
|
||||
switch (a.kind()) {
|
||||
case COMMON_SCHEMA_KIND_INTEGER: {
|
||||
const auto & ia = as<common_schema_integer>(a);
|
||||
const auto & ib = as<common_schema_integer>(b);
|
||||
auto node = std::make_unique<common_schema_integer>();
|
||||
node->minimum = std::max(ia.minimum, ib.minimum);
|
||||
node->maximum = std::min(ia.maximum, ib.maximum);
|
||||
return node->minimum > node->maximum ? none() : std::move(node);
|
||||
}
|
||||
case COMMON_SCHEMA_KIND_STRING: {
|
||||
const auto & sa = as<common_schema_string>(a);
|
||||
const auto & sb = as<common_schema_string>(b);
|
||||
if (!sa.pattern.empty() && !sb.pattern.empty() && sa.pattern != sb.pattern) {
|
||||
return irreducible(a, b);
|
||||
}
|
||||
if (sa.format != COMMON_SCHEMA_FORMAT_NONE && sb.format != COMMON_SCHEMA_FORMAT_NONE && sa.format != sb.format) {
|
||||
return none();
|
||||
}
|
||||
auto node = std::make_unique<common_schema_string>();
|
||||
node->pattern = sa.pattern.empty() ? sb.pattern : sa.pattern;
|
||||
node->format = sa.format == COMMON_SCHEMA_FORMAT_NONE ? sb.format : sa.format;
|
||||
node->min_length = std::max(sa.min_length, sb.min_length);
|
||||
node->max_length = sa.max_length < 0 ? sb.max_length : sb.max_length < 0 ? sa.max_length : std::min(sa.max_length, sb.max_length);
|
||||
return node->max_length >= 0 && node->min_length > node->max_length ? none() : std::move(node);
|
||||
}
|
||||
case COMMON_SCHEMA_KIND_ARRAY: {
|
||||
const auto & aa = as<common_schema_array>(a);
|
||||
const auto & ab = as<common_schema_array>(b);
|
||||
int min_items = std::max(aa.min_items, ab.min_items);
|
||||
int max_items = aa.max_items < 0 ? ab.max_items : ab.max_items < 0 ? aa.max_items : std::min(aa.max_items, ab.max_items);
|
||||
return make_array(intersect(*aa.items, *ab.items), min_items, max_items);
|
||||
}
|
||||
case COMMON_SCHEMA_KIND_TUPLE: {
|
||||
const auto & ta = as<common_schema_tuple>(a);
|
||||
const auto & tb = as<common_schema_tuple>(b);
|
||||
if (ta.items.size() != tb.items.size()) {
|
||||
return none();
|
||||
}
|
||||
auto node = std::make_unique<common_schema_tuple>();
|
||||
for (size_t i = 0; i < ta.items.size(); i++) {
|
||||
auto item = intersect(*ta.items[i], *tb.items[i]);
|
||||
if (is(item, COMMON_SCHEMA_KIND_NONE)) {
|
||||
return none();
|
||||
}
|
||||
node->items.push_back(std::move(item));
|
||||
}
|
||||
return node;
|
||||
}
|
||||
case COMMON_SCHEMA_KIND_OBJECT:
|
||||
return intersect_object(as<common_schema_object>(a), as<common_schema_object>(b));
|
||||
default:
|
||||
// null, boolean and number have no fields to differ in
|
||||
return clone(a);
|
||||
}
|
||||
}
|
||||
|
||||
// Moves the refs the node reaches into live, and points each ref node at its target there.
|
||||
void link(common_schema & node, std::map<std::string, common_schema_ptr> & live) {
|
||||
switch (node.kind()) {
|
||||
case COMMON_SCHEMA_KIND_REF: {
|
||||
auto & ref = as<common_schema_ref>(node);
|
||||
auto it = live.find(ref.ref);
|
||||
if (it == live.end()) {
|
||||
it = live.emplace(ref.ref, std::move(doc_.refs.at(ref.ref))).first;
|
||||
link(*it->second, live);
|
||||
}
|
||||
ref.target = it->second.get();
|
||||
return;
|
||||
}
|
||||
case COMMON_SCHEMA_KIND_ANY_OF:
|
||||
for (auto & child : as<common_schema_any_of>(node).children) {
|
||||
link(*child, live);
|
||||
}
|
||||
return;
|
||||
case COMMON_SCHEMA_KIND_ALL_OF:
|
||||
for (auto & child : as<common_schema_all_of>(node).children) {
|
||||
link(*child, live);
|
||||
}
|
||||
return;
|
||||
case COMMON_SCHEMA_KIND_ARRAY:
|
||||
link(*as<common_schema_array>(node).items, live);
|
||||
return;
|
||||
case COMMON_SCHEMA_KIND_TUPLE:
|
||||
for (auto & item : as<common_schema_tuple>(node).items) {
|
||||
link(*item, live);
|
||||
}
|
||||
return;
|
||||
case COMMON_SCHEMA_KIND_OBJECT: {
|
||||
auto & obj = as<common_schema_object>(node);
|
||||
for (auto & prop : obj.properties) {
|
||||
link(*prop.schema, live);
|
||||
}
|
||||
if (obj.additional_properties) {
|
||||
link(*obj.additional_properties, live);
|
||||
}
|
||||
return;
|
||||
}
|
||||
default:
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
public:
|
||||
explicit common_schema_optimizer(common_schema_document & doc) : doc_(doc) {}
|
||||
|
||||
void run() {
|
||||
// a $ref whose target became none is none too, which the next pass can then prune in its parent
|
||||
do {
|
||||
changed_ = false;
|
||||
for (auto & entry : doc_.refs) {
|
||||
entry.second = rewrite(std::move(entry.second));
|
||||
}
|
||||
doc_.root = rewrite(std::move(doc_.root));
|
||||
} while (changed_);
|
||||
|
||||
// only the refs the root still reaches are kept, and every ref node gets its new target
|
||||
std::map<std::string, common_schema_ptr> live;
|
||||
link(*doc_.root, live);
|
||||
doc_.refs = std::move(live);
|
||||
}
|
||||
};
|
||||
|
||||
void common_schema_optimize(common_schema_document & doc) {
|
||||
common_schema_optimizer(doc).run();
|
||||
}
|
||||
|
||||
static common_schema_kind kind_of(const common_json & value) {
|
||||
if (value.is_null()) {
|
||||
return COMMON_SCHEMA_KIND_NULL;
|
||||
@@ -972,8 +374,6 @@ static common_schema_kinds resolve_kinds(const common_schema & s, std::unordered
|
||||
switch (s.kind()) {
|
||||
case COMMON_SCHEMA_KIND_ANY:
|
||||
return common_schema_kinds::all();
|
||||
case COMMON_SCHEMA_KIND_NONE:
|
||||
return {};
|
||||
case COMMON_SCHEMA_KIND_NUMBER:
|
||||
return { COMMON_SCHEMA_KIND_NUMBER, COMMON_SCHEMA_KIND_INTEGER };
|
||||
case COMMON_SCHEMA_KIND_TUPLE:
|
||||
|
||||
@@ -14,7 +14,6 @@
|
||||
|
||||
enum common_schema_kind {
|
||||
COMMON_SCHEMA_KIND_ANY,
|
||||
COMMON_SCHEMA_KIND_NONE,
|
||||
COMMON_SCHEMA_KIND_REF,
|
||||
COMMON_SCHEMA_KIND_ANY_OF,
|
||||
COMMON_SCHEMA_KIND_ALL_OF,
|
||||
@@ -51,11 +50,6 @@ struct common_schema_any : common_schema {
|
||||
common_schema_kind kind() const override { return COMMON_SCHEMA_KIND_ANY; }
|
||||
};
|
||||
|
||||
// Matches no value: what common_schema_optimize() leaves where an intersection turned out empty
|
||||
struct common_schema_none : common_schema {
|
||||
common_schema_kind kind() const override { return COMMON_SCHEMA_KIND_NONE; }
|
||||
};
|
||||
|
||||
// {"$ref": "#/..."}, only references into the same document are supported
|
||||
struct common_schema_ref : common_schema {
|
||||
std::string ref;
|
||||
@@ -165,12 +159,6 @@ common_schema_document common_schema_parse(const common_json & schema);
|
||||
// doc is unchanged when the schema is rejected.
|
||||
common_schema_ptr common_schema_parse(const common_json & schema, common_schema_document & doc);
|
||||
|
||||
// Rewrites a document in place into an equivalent one with less redundancy: allOf becomes the intersection
|
||||
// of its children, nested anyOf are flattened, branches that can match nothing are pruned, up to a
|
||||
// common_schema_none root, and $refs nothing reaches anymore are dropped.
|
||||
// An allOf stays where the children cannot be combined, e.g. two different patterns or a const against a type.
|
||||
void common_schema_optimize(common_schema_document & doc);
|
||||
|
||||
// A set of the kinds of value a schema may match: only the value kinds NULL to OBJECT occur, a tuple counts as an array
|
||||
class common_schema_kinds {
|
||||
uint32_t mask_ = 0;
|
||||
|
||||
@@ -45,7 +45,7 @@ common_chat_params common_chat_params_init_functionary_v3_2(const common_chat_te
|
||||
foreach_function(inputs.tools, [&](const json & tool) {
|
||||
const auto & function = tool.at("function");
|
||||
std::string name = function.at("name");
|
||||
const auto & schema = function.at("parameters");
|
||||
auto schema = common_chat_tool_parameters(function);
|
||||
|
||||
// Tool format: >>>function_name\n{json_args}
|
||||
auto tool_parser = p.tool(
|
||||
|
||||
@@ -33,7 +33,7 @@ common_chat_params common_chat_params_init_gigachat_v3(
|
||||
for (const auto & tool : inputs.tools) {
|
||||
const auto & function = tool.at("function");
|
||||
std::string name = function.at("name");
|
||||
const auto & schema = function.at("parameters");
|
||||
auto schema = common_chat_tool_parameters(function);
|
||||
|
||||
auto tool_name = p.json_member("name", "\"" + p.tool_name(p.literal(name)) + "\"");
|
||||
auto tool_args = p.json_member("arguments", p.tool_args(p.schema(p.json(), "tool-" + name + "-schema", schema)));
|
||||
|
||||
@@ -109,7 +109,7 @@ common_chat_params common_chat_params_init_gpt_oss(const common_chat_template &
|
||||
foreach_function(inputs.tools, [&](const json & tool) {
|
||||
const auto & function = tool.at("function");
|
||||
std::string name = function.at("name");
|
||||
const auto & params = function.at("parameters");
|
||||
auto params = common_chat_tool_parameters(function);
|
||||
|
||||
auto func_name = p.literal(" to=functions.") + p.tool_name(p.literal(name));
|
||||
auto constraint = p.optional(p.space() + p.optional(p.literal("<|constrain|>")) + constrain_type);
|
||||
|
||||
@@ -82,7 +82,7 @@ common_chat_params common_chat_params_init_kimi_k2(const common_chat_template &
|
||||
foreach_function(inputs.tools, [&](const json & tool) {
|
||||
const auto & function = tool.at("function");
|
||||
std::string name = function.at("name");
|
||||
const auto & schema = function.at("parameters");
|
||||
auto schema = common_chat_tool_parameters(function);
|
||||
|
||||
// Match: functions.<name>:<digits>
|
||||
// Capture the full call id (functions.<name>:<digits>) using tool_id tag
|
||||
|
||||
@@ -89,7 +89,7 @@ common_chat_params common_chat_params_init_ministral_3(const common_chat_templat
|
||||
foreach_function(inputs.tools, [&](const json & tool) {
|
||||
const auto & function = tool.at("function");
|
||||
std::string name = function.at("name");
|
||||
const auto & schema = function.at("parameters");
|
||||
auto schema = common_chat_tool_parameters(function);
|
||||
|
||||
tool_choice |=
|
||||
p.rule("tool-" + name, p.tool_open(p.tool_name(p.literal(name)) + "[ARGS]") +
|
||||
|
||||
@@ -472,6 +472,12 @@ static common_chat_tool empty_args_tool_no_properties{
|
||||
})",
|
||||
};
|
||||
|
||||
static common_chat_tool empty_args_tool_no_schema{
|
||||
/* .name = */ "empty_args_no_schema",
|
||||
/* .description = */ "A tool that takes no arguments and has no parameters schema",
|
||||
/* .parameters = */ "{}",
|
||||
};
|
||||
|
||||
static common_chat_tool python_tool{
|
||||
/* .name = */ "python",
|
||||
/* .description = */ "an ipython interpreter",
|
||||
@@ -5071,6 +5077,13 @@ static void test_template_output_peg_parsers(bool detailed_debug) {
|
||||
.expect(simple_assist_msg("", "", "empty_args", "{}"))
|
||||
.run();
|
||||
|
||||
// Tool call with no parameters schema, {} means no arguments
|
||||
tst.test("<tool_call>\n{\"name\": \"empty_args_no_schema\", \"arguments\": {}}</tool_call>")
|
||||
.enable_thinking(false)
|
||||
.tools({ empty_args_tool_no_schema })
|
||||
.expect(simple_assist_msg("", "", "empty_args_no_schema", "{}"))
|
||||
.run();
|
||||
|
||||
// fake tool call marker in reasoning
|
||||
tst.test(
|
||||
"Let me think about <tool_call>\n{\"name\": \"special_function\", \"arguments\": {\"arg1\": 2}}</tool_call> hmm\n</think>\n\n"
|
||||
|
||||
@@ -49,7 +49,6 @@ static std::string dump_range(int64_t min, int64_t min_def, int64_t max, int64_t
|
||||
static std::string dump(const common_schema & node) {
|
||||
switch (node.kind()) {
|
||||
case COMMON_SCHEMA_KIND_ANY: return "any";
|
||||
case COMMON_SCHEMA_KIND_NONE: return "none";
|
||||
case COMMON_SCHEMA_KIND_NULL: return "null";
|
||||
case COMMON_SCHEMA_KIND_BOOLEAN: return "boolean";
|
||||
case COMMON_SCHEMA_KIND_NUMBER: return "number";
|
||||
@@ -100,12 +99,6 @@ static std::string dump(const common_schema & node) {
|
||||
return "?";
|
||||
}
|
||||
|
||||
static std::string optimize(const std::string & schema) {
|
||||
auto doc = parse(schema);
|
||||
common_schema_optimize(doc);
|
||||
return dump(*doc.root);
|
||||
}
|
||||
|
||||
static void assert_error(testing & t, const std::string & schema, const std::string & needle) {
|
||||
try {
|
||||
parse(schema);
|
||||
@@ -706,296 +699,6 @@ static void test_errors(testing & t) {
|
||||
});
|
||||
}
|
||||
|
||||
static void test_optimize_any_of(testing & t) {
|
||||
t.test("nested anyOf are flattened", [](testing & t) {
|
||||
t.assert_equal("anyOf(string, null, boolean)", optimize(R"({"anyOf": [{"anyOf": [{"type": "string"}, {"type": "null"}]}, {"type": "boolean"}]})"));
|
||||
});
|
||||
|
||||
t.test("one alternative left unwraps", [](testing & t) {
|
||||
t.assert_equal("null", optimize(R"({"oneOf": [{"type": "null"}]})"));
|
||||
});
|
||||
|
||||
t.test("any absorbs the rest", [](testing & t) {
|
||||
t.assert_equal("any", optimize(R"({"anyOf": [{"type": "string"}, {}, {"type": "null"}]})"));
|
||||
});
|
||||
|
||||
t.test("an alternative that matches nothing is pruned", [](testing & t) {
|
||||
t.assert_equal("string", optimize(R"({"anyOf": [{"type": "integer", "minimum": 5, "maximum": 1}, {"type": "string"}]})"));
|
||||
t.assert_equal("none", optimize(R"({"anyOf": [{"type": "integer", "minimum": 5, "maximum": 1}]})"));
|
||||
});
|
||||
}
|
||||
|
||||
static void test_optimize_all_of(testing & t) {
|
||||
t.test("objects merge their properties", [](testing & t) {
|
||||
t.assert_equal("object{a: string, b?: integer}", optimize(R"({"allOf": [
|
||||
{"properties": {"a": {"type": "string"}}, "required": ["a"]},
|
||||
{"properties": {"b": {"type": "integer"}}}
|
||||
]})"));
|
||||
});
|
||||
|
||||
t.test("a shared property is intersected and required by either side", [](testing & t) {
|
||||
t.assert_equal("object{a: integer[1..5]}", optimize(R"({"allOf": [
|
||||
{"properties": {"a": {"type": "integer", "minimum": 1}}, "required": ["a"]},
|
||||
{"properties": {"a": {"type": "integer", "maximum": 5}}}
|
||||
]})"));
|
||||
});
|
||||
|
||||
t.test("additionalProperties constrains the other side's properties", [](testing & t) {
|
||||
t.assert_equal("object{a?: integer[0..]}", optimize(R"({"allOf": [
|
||||
{"properties": {"a": {"type": "integer"}}},
|
||||
{"additionalProperties": {"type": "integer", "minimum": 0}}
|
||||
]})"));
|
||||
t.assert_equal("object{a?: integer, *: integer}", optimize(R"({"allOf": [
|
||||
{"properties": {"a": {"type": "number"}}, "additionalProperties": true},
|
||||
{"additionalProperties": {"type": "integer"}}
|
||||
]})"));
|
||||
t.assert_equal("object{*: integer}", optimize(R"({"allOf": [
|
||||
{"properties": {"a": {"type": "string"}}, "additionalProperties": true},
|
||||
{"additionalProperties": {"type": "integer"}}
|
||||
]})"));
|
||||
});
|
||||
|
||||
t.test("a shared property that matches nothing", [](testing & t) {
|
||||
t.assert_equal("object{b?: any}", optimize(R"({"allOf": [
|
||||
{"properties": {"a": {"type": "string"}, "b": {}}},
|
||||
{"properties": {"a": {"type": "integer"}}}
|
||||
]})"));
|
||||
t.assert_equal("none", optimize(R"({"allOf": [
|
||||
{"properties": {"a": {"type": "string"}}, "required": ["a"]},
|
||||
{"properties": {"a": {"type": "integer"}}}
|
||||
]})"));
|
||||
});
|
||||
|
||||
t.test("a ref is inlined and dropped", [](testing & t) {
|
||||
auto doc = parse(R"({
|
||||
"allOf": [{"$ref": "#/$defs/base"}, {"properties": {"b": {"type": "string"}}}],
|
||||
"$defs": {"base": {"properties": {"a": {"type": "string"}}, "required": ["a"]}}
|
||||
})");
|
||||
common_schema_optimize(doc);
|
||||
t.assert_equal("root", "object{a: string, b?: string}", dump(*doc.root));
|
||||
t.assert_true("no refs", doc.refs.empty());
|
||||
});
|
||||
|
||||
t.test("enums and consts intersect", [](testing & t) {
|
||||
t.assert_equal("const(\"b\")", optimize(R"({"type": "string", "allOf": [{"enum": ["a", "b"]}, {"enum": ["b", "c"]}]})"));
|
||||
t.assert_equal("enum(1, 2)", optimize(R"({"allOf": [{"enum": [1, "x", 2]}, {"enum": [2, 1, 3]}]})"));
|
||||
t.assert_equal("const(1)", optimize(R"({"allOf": [{"const": 1}, {"enum": [2, 1]}]})"));
|
||||
t.assert_equal("none", optimize(R"({"allOf": [{"enum": ["a"]}, {"enum": ["b"]}]})"));
|
||||
t.assert_equal("none", optimize(R"({"allOf": [{"const": 5}, {"const": 6}]})"));
|
||||
});
|
||||
|
||||
t.test("a value against a type stays an allOf", [](testing & t) {
|
||||
t.assert_equal("allOf(const(5), integer)", optimize(R"({"allOf": [{"const": 5}, {"type": "integer"}]})"));
|
||||
});
|
||||
|
||||
t.test("integer bounds", [](testing & t) {
|
||||
t.assert_equal("integer[1..10]", optimize(R"({"allOf": [{"type": "integer", "minimum": 1}, {"type": "integer", "maximum": 10}]})"));
|
||||
t.assert_equal("integer[3..5]", optimize(R"({"allOf": [{"type": "integer", "minimum": 1, "maximum": 5}, {"type": "integer", "minimum": 3, "maximum": 10}]})"));
|
||||
t.assert_equal("none", optimize(R"({"allOf": [{"type": "integer", "maximum": 1}, {"type": "integer", "minimum": 2}]})"));
|
||||
t.assert_equal("integer", optimize(R"({"allOf": [{"type": "number"}, {"type": "integer"}]})"));
|
||||
});
|
||||
|
||||
t.test("strings", [](testing & t) {
|
||||
t.assert_equal("string:date[2..5]", optimize(R"({"type": "string", "allOf": [{"minLength": 2, "type": "string"}, {"type": "string", "maxLength": 5, "format": "date"}]})"));
|
||||
t.assert_equal("string/^a/[1..3]", optimize(R"({"allOf": [{"pattern": "^a", "maxLength": 3}, {"pattern": "^a", "minLength": 1}]})"));
|
||||
t.assert_equal("allOf(string/^a/, string/b$/)", optimize(R"({"allOf": [{"pattern": "^a"}, {"pattern": "b$"}]})"));
|
||||
t.assert_equal("none", optimize(R"({"allOf": [{"format": "date"}, {"format": "uuid"}]})"));
|
||||
t.assert_equal("none", optimize(R"({"allOf": [{"type": "string", "minLength": 5}, {"type": "string", "maxLength": 2}]})"));
|
||||
});
|
||||
|
||||
t.test("kinds that cannot both hold", [](testing & t) {
|
||||
t.assert_equal("none", optimize(R"({"allOf": [{"type": "string"}, {"type": "integer"}]})"));
|
||||
t.assert_equal("none", optimize(R"({"allOf": [{"type": "object"}, {"items": {}}]})"));
|
||||
});
|
||||
|
||||
t.test("any and equal children fold away", [](testing & t) {
|
||||
t.assert_equal("boolean", optimize(R"({"allOf": [{}, {"type": "boolean"}, {"type": "boolean"}]})"));
|
||||
t.assert_equal("any", optimize(R"({"allOf": [{}, {}]})"));
|
||||
});
|
||||
|
||||
t.test("arrays and tuples", [](testing & t) {
|
||||
t.assert_equal("array(integer[0..])[1..3]", optimize(R"({"allOf": [
|
||||
{"items": {"type": "integer"}, "minItems": 1},
|
||||
{"items": {"type": "integer", "minimum": 0}, "maxItems": 3}
|
||||
]})"));
|
||||
t.assert_equal("tuple()", optimize(R"({"allOf": [{"items": {"type": "string"}}, {"items": {"type": "integer"}}]})"));
|
||||
t.assert_equal("tuple(integer[1..], null)", optimize(R"({"allOf": [{"prefixItems": [{"type": "integer"}, {}]}, {"prefixItems": [{"minimum": 1, "type": "integer"}, {"type": "null"}]}]})"));
|
||||
t.assert_equal("none", optimize(R"({"allOf": [{"prefixItems": [{}]}, {"prefixItems": [{}, {}]}]})"));
|
||||
t.assert_equal("allOf(array(integer), tuple(integer))", optimize(R"({"allOf": [{"items": {"type": "integer"}}, {"prefixItems": [{"type": "integer"}]}]})"));
|
||||
});
|
||||
|
||||
t.test("distributes over anyOf", [](testing & t) {
|
||||
t.assert_equal("integer[0..]", optimize(R"({"allOf": [
|
||||
{"anyOf": [{"type": "string"}, {"type": "integer"}]},
|
||||
{"type": "integer", "minimum": 0}
|
||||
]})"));
|
||||
t.assert_equal("anyOf(object{a?: string, c: integer}, object{b?: string, c: integer})", optimize(R"({"allOf": [
|
||||
{"anyOf": [{"properties": {"a": {"type": "string"}}}, {"properties": {"b": {"type": "string"}}}]},
|
||||
{"properties": {"c": {"type": "integer"}}, "required": ["c"]}
|
||||
]})"));
|
||||
t.assert_equal("anyOf(integer[..3], integer[5..])", optimize(R"({"allOf": [
|
||||
{"anyOf": [{"type": "integer", "maximum": 3}, {"type": "integer", "minimum": 5}]},
|
||||
{"anyOf": [{"type": "integer"}, {"type": "string"}]}
|
||||
]})"));
|
||||
});
|
||||
|
||||
t.test("nested allOf are flattened", [](testing & t) {
|
||||
t.assert_equal("integer[1..3]", optimize(R"({"allOf": [{"allOf": [{"type": "integer", "minimum": 1}]}, {"type": "integer", "maximum": 3}]})"));
|
||||
});
|
||||
|
||||
t.test("what cannot combine keeps the rest merged", [](testing & t) {
|
||||
t.assert_equal("allOf(string/^a/[..5], string/b$/)", optimize(R"({"allOf": [{"pattern": "^a"}, {"pattern": "b$"}, {"type": "string", "maxLength": 5}]})"));
|
||||
});
|
||||
|
||||
t.test("recursive refs do not loop", [](testing & t) {
|
||||
auto doc = parse(R"({
|
||||
"allOf": [{"$ref": "#/$defs/a"}, {"$ref": "#/$defs/b"}],
|
||||
"$defs": {
|
||||
"a": {"properties": {"next": {"$ref": "#/$defs/a"}}},
|
||||
"b": {"properties": {"next": {"$ref": "#/$defs/b"}}}
|
||||
}
|
||||
})");
|
||||
common_schema_optimize(doc);
|
||||
t.assert_equal("root", "object{next?: allOf(ref(#/$defs/a), ref(#/$defs/b))}", dump(*doc.root));
|
||||
t.assert_equal("refs", (size_t) 2, doc.refs.size());
|
||||
});
|
||||
}
|
||||
|
||||
static void test_optimize_prune(testing & t) {
|
||||
t.test("bounds that leave nothing", [](testing & t) {
|
||||
t.assert_equal("none", optimize(R"({"type": "integer", "minimum": 5, "maximum": 1})"));
|
||||
t.assert_equal("none", optimize(R"({"type": "string", "minLength": 5, "maxLength": 1})"));
|
||||
t.assert_equal("none", optimize(R"({"type": "array", "minItems": 5, "maxItems": 1})"));
|
||||
t.assert_equal("integer[1..1]", optimize(R"({"type": "integer", "minimum": 1, "maximum": 1})"));
|
||||
});
|
||||
|
||||
t.test("an optional property that can never be present is dropped", [](testing & t) {
|
||||
t.assert_equal("object{b?: any}", optimize(R"({"properties": {"a": {"type": "integer", "minimum": 5, "maximum": 1}, "b": {}}})"));
|
||||
});
|
||||
|
||||
t.test("a required property that can never be present empties the object", [](testing & t) {
|
||||
t.assert_equal("none", optimize(R"({"properties": {"a": {"type": "integer", "minimum": 5, "maximum": 1}}, "required": ["a"]})"));
|
||||
t.assert_equal("none", optimize(R"({"properties": {"o": {"properties": {"a": {"type": "integer", "minimum": 5, "maximum": 1}}, "required": ["a"]}}, "required": ["o"]})"));
|
||||
});
|
||||
|
||||
t.test("additionalProperties that can never be present closes the object", [](testing & t) {
|
||||
t.assert_equal("object{a?: string}", optimize(R"({"properties": {"a": {"type": "string"}}, "additionalProperties": {"type": "integer", "minimum": 5, "maximum": 1}})"));
|
||||
});
|
||||
|
||||
t.test("items that can never be present leave the empty array", [](testing & t) {
|
||||
t.assert_equal("tuple()", optimize(R"({"items": {"type": "integer", "minimum": 5, "maximum": 1}})"));
|
||||
t.assert_equal("tuple()", optimize(R"({"type": "array", "maxItems": 0})"));
|
||||
t.assert_equal("none", optimize(R"({"items": {"type": "integer", "minimum": 5, "maximum": 1}, "minItems": 1})"));
|
||||
});
|
||||
|
||||
t.test("a tuple item that can never be present", [](testing & t) {
|
||||
t.assert_equal("none", optimize(R"({"prefixItems": [{"type": "string"}, {"type": "integer", "minimum": 5, "maximum": 1}]})"));
|
||||
});
|
||||
|
||||
t.test("enum values dedupe", [](testing & t) {
|
||||
t.assert_equal("enum(\"a\", \"b\")", optimize(R"({"enum": ["a", "a", "b"]})"));
|
||||
t.assert_equal("const(\"a\")", optimize(R"({"enum": ["a", "a"]})"));
|
||||
});
|
||||
|
||||
t.test("nested pruning reaches the root", [](testing & t) {
|
||||
t.assert_equal("none", optimize(R"({"anyOf": [
|
||||
{"properties": {"a": {"allOf": [{"type": "string"}, {"type": "null"}]}}, "required": ["a"]},
|
||||
{"prefixItems": [{"allOf": [{"const": 1}, {"const": 2}]}]}
|
||||
]})"));
|
||||
});
|
||||
|
||||
t.test("what is already minimal is left alone", [](testing & t) {
|
||||
auto doc = parse(R"({
|
||||
"properties": {
|
||||
"name": {"type": "string", "minLength": 1},
|
||||
"tags": {"type": "array", "items": {"enum": ["a", "b"]}, "maxItems": 3},
|
||||
"kind": {"anyOf": [{"type": "null"}, {"$ref": "#/$defs/kind"}]}
|
||||
},
|
||||
"required": ["name"],
|
||||
"$defs": {"kind": {"properties": {"id": {"type": "integer", "minimum": 0}}, "required": ["id"]}}
|
||||
})");
|
||||
std::string before = dump(*doc.root);
|
||||
common_schema_optimize(doc);
|
||||
t.assert_equal("root", before, dump(*doc.root));
|
||||
t.assert_equal("root", "object{name: string[1..], tags?: array(enum(\"a\", \"b\"))[..3], kind?: anyOf(null, ref(#/$defs/kind))}", dump(*doc.root));
|
||||
t.assert_equal("refs", (size_t) 1, doc.refs.size());
|
||||
});
|
||||
}
|
||||
|
||||
static void test_optimize_ref(testing & t) {
|
||||
t.test("a ref to nothing is nothing", [](testing & t) {
|
||||
auto doc = parse(R"({"$ref": "#/$defs/t", "$defs": {"t": {"type": "integer", "minimum": 5, "maximum": 1}}})");
|
||||
common_schema_optimize(doc);
|
||||
t.assert_equal("root", "none", dump(*doc.root));
|
||||
t.assert_true("no refs", doc.refs.empty());
|
||||
});
|
||||
|
||||
t.test("a branch through a ref to nothing is pruned", [](testing & t) {
|
||||
auto doc = parse(R"({
|
||||
"anyOf": [{"properties": {"x": {"$ref": "#/$defs/t"}}, "required": ["x"]}, {"type": "string"}],
|
||||
"$defs": {"t": {"allOf": [{"type": "string"}, {"type": "null"}]}}
|
||||
})");
|
||||
common_schema_optimize(doc);
|
||||
t.assert_equal("root", "string", dump(*doc.root));
|
||||
t.assert_true("no refs", doc.refs.empty());
|
||||
});
|
||||
|
||||
t.test("a chain of refs to nothing", [](testing & t) {
|
||||
auto doc = parse(R"({
|
||||
"properties": {"a": {"$ref": "#/$defs/a"}, "b": {}},
|
||||
"$defs": {"a": {"$ref": "#/$defs/b"}, "b": {"$ref": "#/$defs/c"}, "c": {"type": "integer", "minimum": 5, "maximum": 1}}
|
||||
})");
|
||||
common_schema_optimize(doc);
|
||||
t.assert_equal("root", "object{b?: any}", dump(*doc.root));
|
||||
t.assert_true("no refs", doc.refs.empty());
|
||||
});
|
||||
|
||||
t.test("reachable refs are kept and relinked", [](testing & t) {
|
||||
auto doc = parse(R"({
|
||||
"properties": {"a": {"$ref": "#/$defs/t"}, "b": {"$ref": "#/$defs/t"}, "c": {"$ref": "#/$defs/u"}},
|
||||
"$defs": {"t": {"anyOf": [{"type": "string"}]}, "u": {"type": "null"}, "unused": {"type": "boolean"}}
|
||||
})");
|
||||
common_schema_optimize(doc);
|
||||
t.assert_equal("root", "object{a?: ref(#/$defs/t), b?: ref(#/$defs/t), c?: ref(#/$defs/u)}", dump(*doc.root));
|
||||
t.assert_equal("refs", (size_t) 2, doc.refs.size());
|
||||
t.assert_equal("t", "string", dump(*doc.refs.at("#/$defs/t")));
|
||||
const auto & o = root<common_schema_object>(t, doc);
|
||||
for (const auto & prop : o.properties) {
|
||||
const auto & r = as<common_schema_ref>(t, prop.schema.get(), prop.name);
|
||||
t.assert_true(prop.name + " target", r.target != nullptr && r.target == doc.refs.at(r.ref).get());
|
||||
}
|
||||
});
|
||||
|
||||
t.test("a recursive schema survives", [](testing & t) {
|
||||
auto doc = parse(R"({
|
||||
"$ref": "#/$defs/node",
|
||||
"$defs": {
|
||||
"node": {
|
||||
"properties": {
|
||||
"value": {"type": "number"},
|
||||
"next": {"anyOf": [{"$ref": "#/$defs/node"}, {"type": "null"}]}
|
||||
},
|
||||
"required": ["value"]
|
||||
}
|
||||
}
|
||||
})");
|
||||
common_schema_optimize(doc);
|
||||
const auto & r = root<common_schema_ref>(t, doc);
|
||||
t.assert_equal("node", "object{value: number, next?: anyOf(ref(#/$defs/node), null)}", dump(*r.target));
|
||||
t.assert_true("target", r.target == doc.refs.at("#/$defs/node").get());
|
||||
const auto & node = as<common_schema_object>(t, r.target, "node");
|
||||
const auto & next = as<common_schema_any_of>(t, node.properties[1].schema.get(), "next");
|
||||
t.assert_true("cycle", as<common_schema_ref>(t, next.children[0].get(), "next[0]").target == r.target);
|
||||
});
|
||||
|
||||
t.test("a ref intersected with any or itself stays a ref", [](testing & t) {
|
||||
auto doc = parse(R"({"allOf": [{}, {"$ref": "#/$defs/t"}, {"$ref": "#/$defs/t"}], "$defs": {"t": {"type": "boolean"}}})");
|
||||
common_schema_optimize(doc);
|
||||
t.assert_equal("root", "ref(#/$defs/t)", dump(*doc.root));
|
||||
t.assert_true("target", root<common_schema_ref>(t, doc).target == doc.refs.at("#/$defs/t").get());
|
||||
});
|
||||
}
|
||||
|
||||
int main(int argc, char * argv[]) {
|
||||
testing t(std::cout);
|
||||
if (argc >= 2) {
|
||||
@@ -1019,10 +722,6 @@ int main(int argc, char * argv[]) {
|
||||
t.test("all_of", test_all_of);
|
||||
t.test("ref", test_ref);
|
||||
t.test("errors", test_errors);
|
||||
t.test("optimize any_of", test_optimize_any_of);
|
||||
t.test("optimize all_of", test_optimize_all_of);
|
||||
t.test("optimize prune", test_optimize_prune);
|
||||
t.test("optimize ref", test_optimize_ref);
|
||||
|
||||
return t.summary();
|
||||
}
|
||||
|
||||
+2
-2
@@ -133,8 +133,8 @@
|
||||
| `-l, --logit-bias TOKEN_ID(+/-)BIAS` | modifies the likelihood of token appearing in the completion,<br/>i.e. `--logit-bias 15043+1` to increase likelihood of token ' Hello',<br/>or `--logit-bias 15043-1` to decrease likelihood of token ' Hello' |
|
||||
| `--grammar GRAMMAR` | BNF-like grammar to constrain generations (see samples in grammars/ dir) |
|
||||
| `--grammar-file FNAME` | file to read grammar from |
|
||||
| `-j, --json-schema SCHEMA` | JSON schema to constrain generations (https://json-schema.org/), e.g. `{}` for any JSON object<br/>For schemas w/ external $refs, use --grammar + example/json_schema_to_grammar.py instead |
|
||||
| `-jf, --json-schema-file FILE` | File containing a JSON schema to constrain generations (https://json-schema.org/), e.g. `{}` for any JSON object<br/>For schemas w/ external $refs, use --grammar + example/json_schema_to_grammar.py instead |
|
||||
| `-j, --json-schema SCHEMA` | JSON schema to constrain generations (https://json-schema.org/), e.g. `{"type": "object"}` for any JSON object |
|
||||
| `-jf, --json-schema-file FILE` | File containing a JSON schema to constrain generations (https://json-schema.org/), e.g. `{"type": "object"}` for any JSON object |
|
||||
| `-bs, --backend-sampling` | enable backend sampling (experimental) (default: disabled)<br/>(env: LLAMA_ARG_BACKEND_SAMPLING) |
|
||||
|
||||
|
||||
|
||||
@@ -216,8 +216,8 @@ llama-completion.exe -m models\gemma-1.1-7b-it.Q4_K_M.gguf --ignore-eos -n -1
|
||||
| `-l, --logit-bias TOKEN_ID(+/-)BIAS` | modifies the likelihood of token appearing in the completion,<br/>i.e. `--logit-bias 15043+1` to increase likelihood of token ' Hello',<br/>or `--logit-bias 15043-1` to decrease likelihood of token ' Hello' |
|
||||
| `--grammar GRAMMAR` | BNF-like grammar to constrain generations (see samples in grammars/ dir) |
|
||||
| `--grammar-file FNAME` | file to read grammar from |
|
||||
| `-j, --json-schema SCHEMA` | JSON schema to constrain generations (https://json-schema.org/), e.g. `{}` for any JSON object<br/>For schemas w/ external $refs, use --grammar + example/json_schema_to_grammar.py instead |
|
||||
| `-jf, --json-schema-file FILE` | File containing a JSON schema to constrain generations (https://json-schema.org/), e.g. `{}` for any JSON object<br/>For schemas w/ external $refs, use --grammar + example/json_schema_to_grammar.py instead |
|
||||
| `-j, --json-schema SCHEMA` | JSON schema to constrain generations (https://json-schema.org/), e.g. `{"type": "object"}` for any JSON object |
|
||||
| `-jf, --json-schema-file FILE` | File containing a JSON schema to constrain generations (https://json-schema.org/), e.g. `{"type": "object"}` for any JSON object |
|
||||
| `-bs, --backend-sampling` | enable backend sampling (experimental) (default: disabled)<br/>(env: LLAMA_ARG_BACKEND_SAMPLING) |
|
||||
|
||||
|
||||
@@ -556,7 +556,7 @@ These options help improve the performance and memory usage of the LLaMA models.
|
||||
|
||||
- `--grammar GRAMMAR`, `--grammar-file FILE`: Specify a grammar (defined inline or in a file) to constrain model output to a specific format. For example, you could force the model to output JSON or to speak only in emojis. See the [GBNF guide](../../grammars/README.md) for details on the syntax.
|
||||
|
||||
- `--json-schema SCHEMA`: Specify a [JSON schema](https://json-schema.org/) to constrain model output to (e.g. `{}` for any JSON object, or `{"items": {"type": "string", "minLength": 10, "maxLength": 100}, "minItems": 10}` for a JSON array of strings with size constraints). If a schema uses external `$ref`s, you should use `--grammar "$( python examples/json_schema_to_grammar.py myschema.json )"` instead.
|
||||
- `--json-schema SCHEMA`: Specify a [JSON schema](https://json-schema.org/) to constrain model output to (e.g. `{"type": "object"}` for any JSON object, or `{"items": {"type": "string", "minLength": 10, "maxLength": 100}, "minItems": 10}` for a JSON array of strings with size constraints).
|
||||
|
||||
### Quantization
|
||||
|
||||
|
||||
@@ -150,8 +150,8 @@ For the full list of features, please refer to [server's changelog](https://gith
|
||||
| `-l, --logit-bias TOKEN_ID(+/-)BIAS` | modifies the likelihood of token appearing in the completion,<br/>i.e. `--logit-bias 15043+1` to increase likelihood of token ' Hello',<br/>or `--logit-bias 15043-1` to decrease likelihood of token ' Hello' |
|
||||
| `--grammar GRAMMAR` | BNF-like grammar to constrain generations (see samples in grammars/ dir) |
|
||||
| `--grammar-file FNAME` | file to read grammar from |
|
||||
| `-j, --json-schema SCHEMA` | JSON schema to constrain generations (https://json-schema.org/), e.g. `{}` for any JSON object<br/>For schemas w/ external $refs, use --grammar + example/json_schema_to_grammar.py instead |
|
||||
| `-jf, --json-schema-file FILE` | File containing a JSON schema to constrain generations (https://json-schema.org/), e.g. `{}` for any JSON object<br/>For schemas w/ external $refs, use --grammar + example/json_schema_to_grammar.py instead |
|
||||
| `-j, --json-schema SCHEMA` | JSON schema to constrain generations (https://json-schema.org/), e.g. `{"type": "object"}` for any JSON object |
|
||||
| `-jf, --json-schema-file FILE` | File containing a JSON schema to constrain generations (https://json-schema.org/), e.g. `{"type": "object"}` for any JSON object |
|
||||
| `-bs, --backend-sampling` | enable backend sampling (experimental) (default: disabled)<br/>(env: LLAMA_ARG_BACKEND_SAMPLING) |
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user