diff --git a/.github/workflows/build-cpu.yml b/.github/workflows/build-cpu.yml index 9e92314bc2..ddda55f1c6 100644 --- a/.github/workflows/build-cpu.yml +++ b/.github/workflows/build-cpu.yml @@ -221,7 +221,6 @@ jobs: # 7z x "-o${env:RUNNER_TEMP}" $env:RUNNER_TEMP/sde.tar # $sde = $(join-path $env:RUNNER_TEMP sde-external-${env:SDE_VERSION}-win/sde.exe) # cd build - # $env:LLAMA_SKIP_TESTS_SLOW_ON_EMULATOR = 1 # & $sde -future -- ctest -L main -C Release --verbose --timeout 900 - name: ccache-clear diff --git a/common/chat-auto-parser-generator.cpp b/common/chat-auto-parser-generator.cpp index d7e117e4d9..04f98071ab 100644 --- a/common/chat-auto-parser-generator.cpp +++ b/common/chat-auto-parser-generator.cpp @@ -312,7 +312,7 @@ common_peg_parser analyze_tools::build_tool_parser_tag_json(parser_build_context foreach_function(inputs.tools, [&](const json & tool) { const auto & func = tool.at("function"); std::string name = func.at("name"); - const auto & schema = func.contains("parameters") ? func.at("parameters") : json::object(); + auto schema = common_chat_tool_parameters(func); // Build call_id parser based on position (if supported) bool have_call_id = false; diff --git a/common/chat-peg-parser.cpp b/common/chat-peg-parser.cpp index 79b97a80f1..b8e5adc3b2 100644 --- a/common/chat-peg-parser.cpp +++ b/common/chat-peg-parser.cpp @@ -640,7 +640,7 @@ common_peg_parser common_chat_peg_builder::build_json_tools_function_is_key( } const auto & function = tool_def.at("function"); std::string name = function.at("name"); - ordered_json params = function.contains("parameters") ? function.at("parameters") : ordered_json::object(); + ordered_json params = common_chat_tool_parameters(function); // Build inner object fields std::vector inner_fields; @@ -726,7 +726,7 @@ common_peg_parser common_chat_peg_builder::build_json_tools_nested_keys( } const auto & function = tool_def.at("function"); std::string name = function.at("name"); - ordered_json params = function.contains("parameters") ? function.at("parameters") : ordered_json::object(); + ordered_json params = common_chat_tool_parameters(function); auto nested_name = literal("\"" + nested_name_field + "\"") + space() + literal(":") + space() + atomic(literal("\"") + tool_name(literal(name)) + literal("\"")); @@ -795,7 +795,7 @@ common_peg_parser common_chat_peg_builder::build_json_tools_flat_keys( } const auto & function = tool_def.at("function"); std::string name = function.at("name"); - ordered_json params = function.contains("parameters") ? function.at("parameters") : ordered_json::object(); + ordered_json params = common_chat_tool_parameters(function); auto tool_name_ = name_key_parser + space() + literal(":") + space() + atomic(literal("\"") + tool_name(literal(name)) + literal("\"")); @@ -1230,3 +1230,14 @@ void common_chat_peg_minimax_m3_mapper::visit(const common_peg_ast_arena & arena visit(arena, child_id); } } + +common_json common_chat_tool_parameters(const common_json & function) { + if (function.contains("parameters") && function.at("parameters").is_object() && !function.at("parameters").empty()) { + return function.at("parameters"); + } + auto schema = common_json::object(); + schema["type"] = "object"; + schema["properties"] = common_json::object(); + schema["additionalProperties"] = false; + return schema; +} diff --git a/common/chat-peg-parser.h b/common/chat-peg-parser.h index 114fa049fa..06df5a94de 100644 --- a/common/chat-peg-parser.h +++ b/common/chat-peg-parser.h @@ -218,3 +218,7 @@ struct tagged_peg_parser { tagged_peg_parser build_tagged_peg_parser( const std::function & fn); + +// The parameters schema of a tool for its arguments parser. Like the OpenAI API, a missing or empty +// "parameters" means the tool takes no arguments, not that any value is accepted. +common_json common_chat_tool_parameters(const common_json & function); diff --git a/common/json-schema-to-grammar.cpp b/common/json-schema-to-grammar.cpp index 459fe9a797..347b98d6b1 100644 --- a/common/json-schema-to-grammar.cpp +++ b/common/json-schema-to-grammar.cpp @@ -347,7 +347,6 @@ private: std::vector _errors; std::vector _warnings; - // every schema given to resolve_refs() or add_schema() is parsed into this document, so their $refs resolve through each other common_schema_document _doc; template @@ -681,19 +680,19 @@ private: return out.str(); } - std::string _resolve_ref(const common_schema_ref & ref) { - auto it = ref.ref.find('#'); - std::string ref_fragment = it != std::string::npos ? ref.ref.substr(it + 1) : ref.ref; + std::string _resolve_ref(const common_schema_ref & schema) { + auto it = schema.ref.find('#'); + std::string ref_fragment = it != std::string::npos ? schema.ref.substr(it + 1) : schema.ref; static const std::regex nonalphanumeric_regex(R"([^a-zA-Z0-9-]+)"); std::string ref_name = "ref" + std::regex_replace(ref_fragment, nonalphanumeric_regex, "-"); - if (_rules.find(ref_name) == _rules.end() && _refs_being_resolved.find(ref.ref) == _refs_being_resolved.end()) { - if (!ref.target) { - _errors.push_back("Unresolved $ref " + ref.ref); + if (_rules.find(ref_name) == _rules.end() && _refs_being_resolved.find(schema.ref) == _refs_being_resolved.end()) { + if (!schema.target) { + _errors.push_back("Unresolved $ref " + schema.ref); return ""; } - _refs_being_resolved.insert(ref.ref); - ref_name = visit(*ref.target, ref_name); - _refs_being_resolved.erase(ref.ref); + _refs_being_resolved.insert(schema.ref); + ref_name = visit(*schema.target, ref_name); + _refs_being_resolved.erase(schema.ref); } return ref_name; } @@ -848,7 +847,7 @@ public: return _add_primitive(rule_name == "root" ? "root" : type, PRIMITIVE_RULES.at(type)); } - // An allOf merged the way the Python converter does it: the properties of a direct component are required, those of a nested anyOf are optional, enums are intersected. + // An allOf merges its components: the properties of a direct component are required, those of a nested anyOf are optional, enums are intersected. std::string _visit_all_of(const common_schema_all_of & schema, const std::string & name, const std::string & rule_name) { std::unordered_set required; std::vector> properties; @@ -988,9 +987,6 @@ public: return _visit_primitive(rule_name, "null"); case COMMON_SCHEMA_KIND_ANY: return _add_rule(rule_name, _add_primitive("value", PRIMITIVE_RULES.at("value"))); - case COMMON_SCHEMA_KIND_NONE: - _errors.push_back("No value satisfies the schema of " + rule_name); - return ""; } return ""; } diff --git a/common/json-schema.cpp b/common/json-schema.cpp index 6882127409..1aa67c9eb6 100644 --- a/common/json-schema.cpp +++ b/common/json-schema.cpp @@ -1,10 +1,8 @@ #include "json-schema.h" #include "common.h" -#include #include #include -#include #include #include #include @@ -350,602 +348,6 @@ common_schema_ptr common_schema_parse(const common_json & schema, common_schema_ return common_schema_parser(schema, doc).parse(); } -class common_schema_optimizer { - common_schema_document & doc_; - - // set when a $ref got replaced by the none its target became, the pass is then repeated - bool changed_ = false; - - // the target pairs being intersected further up, a recursive schema would otherwise never bottom out - std::set> active_; - - template - static const T & as(const common_schema & node) { - return static_cast(node); - } - - template - static T & as(common_schema & node) { - return static_cast(node); - } - - static bool is(const common_schema & node, common_schema_kind kind) { - return node.kind() == kind; - } - - static bool is(const common_schema_ptr & node, common_schema_kind kind) { - return node && node->kind() == kind; - } - - static common_schema_ptr none() { - return std::make_unique(); - } - - // Follows a chain of $refs to a node of another kind. - // Gives nullptr for a target that is being rewritten right now, or a chain that loops back on itself. - const common_schema * resolve(const common_schema & node) const { - const common_schema * cur = &node; - for (size_t hops = 0; is(*cur, COMMON_SCHEMA_KIND_REF); hops++) { - if (hops > doc_.refs.size()) { - return nullptr; - } - auto it = doc_.refs.find(as(*cur).ref); - if (it == doc_.refs.end() || !it->second) { - return nullptr; - } - cur = it->second.get(); - } - return cur; - } - - static std::vector clone_all(const std::vector & nodes) { - std::vector out; - for (const auto & node : nodes) { - out.push_back(clone(*node)); - } - return out; - } - - static common_schema_ptr clone(const common_schema & node) { - switch (node.kind()) { - case COMMON_SCHEMA_KIND_ANY: return std::make_unique(); - case COMMON_SCHEMA_KIND_NONE: return none(); - case COMMON_SCHEMA_KIND_NULL: return std::make_unique(); - case COMMON_SCHEMA_KIND_BOOLEAN: return std::make_unique(); - case COMMON_SCHEMA_KIND_NUMBER: return std::make_unique(); - case COMMON_SCHEMA_KIND_INTEGER: return std::make_unique(as(node)); - case COMMON_SCHEMA_KIND_STRING: return std::make_unique(as(node)); - case COMMON_SCHEMA_KIND_CONST: return std::make_unique(as(node).value); - case COMMON_SCHEMA_KIND_REF: { - const auto & ref = as(node); - auto out = std::make_unique(ref.ref); - out->target = ref.target; - return out; - } - case COMMON_SCHEMA_KIND_ENUM: { - auto out = std::make_unique(); - out->values = as(node).values; - return out; - } - case COMMON_SCHEMA_KIND_ANY_OF: { - auto out = std::make_unique(); - out->children = clone_all(as(node).children); - return out; - } - case COMMON_SCHEMA_KIND_ALL_OF: { - auto out = std::make_unique(); - out->children = clone_all(as(node).children); - return out; - } - case COMMON_SCHEMA_KIND_ARRAY: { - const auto & arr = as(node); - auto out = std::make_unique(); - out->items = clone(*arr.items); - out->min_items = arr.min_items; - out->max_items = arr.max_items; - return out; - } - case COMMON_SCHEMA_KIND_TUPLE: { - auto out = std::make_unique(); - out->items = clone_all(as(node).items); - return out; - } - case COMMON_SCHEMA_KIND_OBJECT: { - const auto & obj = as(node); - auto out = std::make_unique(); - for (const auto & prop : obj.properties) { - out->properties.push_back({prop.name, clone(*prop.schema), prop.required}); - } - if (obj.additional_properties) { - out->additional_properties = clone(*obj.additional_properties); - } - return out; - } - } - return none(); - } - - static bool contains(const std::vector & values, const common_json & value) { - return std::any_of(values.begin(), values.end(), [&](const common_json & v) { return v == value; }); - } - - // the values a const or enum accepts - static bool get_values(const common_schema & node, std::vector & values) { - if (is(node, COMMON_SCHEMA_KIND_CONST)) { - values.push_back(as(node).value); - return true; - } - if (is(node, COMMON_SCHEMA_KIND_ENUM)) { - values = as(node).values; - return true; - } - return false; - } - - // a const for one value, none for no value at all - static common_schema_ptr make_values(std::vector values) { - if (values.empty()) { - return none(); - } - if (values.size() == 1) { - return std::make_unique(std::move(values[0])); - } - auto node = std::make_unique(); - node->values = std::move(values); - return node; - } - - // The union of rewritten alternatives. - static common_schema_ptr make_any_of(std::vector children) { - std::vector flat; - for (auto & child : children) { - if (is(child, COMMON_SCHEMA_KIND_ANY_OF)) { - // a nested one is flat already - for (auto & c : as(*child).children) { - flat.push_back(std::move(c)); - } - } else { - flat.push_back(std::move(child)); - } - } - std::vector out; - for (auto & child : flat) { - if (is(child, COMMON_SCHEMA_KIND_ANY)) { - return std::make_unique(); - } - if (!is(child, COMMON_SCHEMA_KIND_NONE)) { - out.push_back(std::move(child)); - } - } - if (out.empty()) { - return none(); - } - if (out.size() == 1) { - return std::move(out[0]); - } - auto node = std::make_unique(); - node->children = std::move(out); - return node; - } - - // An array with rewritten items. - static common_schema_ptr make_array(common_schema_ptr items, int min_items, int max_items) { - if ((max_items >= 0 && min_items > max_items) || (is(items, COMMON_SCHEMA_KIND_NONE) && min_items > 0)) { - return none(); - } - if (max_items == 0 || is(items, COMMON_SCHEMA_KIND_NONE)) { - // only [] is left - return std::make_unique(); - } - auto node = std::make_unique(); - node->items = std::move(items); - node->min_items = min_items; - node->max_items = max_items; - return node; - } - - // An object with rewritten property schemas. - static common_schema_ptr make_object(std::vector properties, common_schema_ptr additional) { - auto node = std::make_unique(); - for (auto & prop : properties) { - if (is(prop.schema, COMMON_SCHEMA_KIND_NONE)) { - // it can never be present - if (prop.required) { - return none(); - } - continue; - } - node->properties.push_back(std::move(prop)); - } - if (additional && !is(additional, COMMON_SCHEMA_KIND_NONE)) { - node->additional_properties = std::move(additional); - } - return node; - } - - std::vector rewrite_all(std::vector nodes) { - for (auto & node : nodes) { - node = rewrite(std::move(node)); - } - return nodes; - } - - // Rewrites a node bottom-up. - common_schema_ptr rewrite(common_schema_ptr node) { - switch (node->kind()) { - case COMMON_SCHEMA_KIND_ANY: - case COMMON_SCHEMA_KIND_NONE: - case COMMON_SCHEMA_KIND_CONST: - case COMMON_SCHEMA_KIND_NULL: - case COMMON_SCHEMA_KIND_BOOLEAN: - case COMMON_SCHEMA_KIND_NUMBER: - return node; - case COMMON_SCHEMA_KIND_REF: { - const common_schema * target = resolve(*node); - if (target && is(*target, COMMON_SCHEMA_KIND_NONE)) { - changed_ = true; - return none(); - } - return node; - } - case COMMON_SCHEMA_KIND_ENUM: { - std::vector values; - for (const auto & v : as(*node).values) { - if (!contains(values, v)) { - values.push_back(v); - } - } - return make_values(std::move(values)); - } - case COMMON_SCHEMA_KIND_INTEGER: { - const auto & i = as(*node); - return i.minimum > i.maximum ? none() : std::move(node); - } - case COMMON_SCHEMA_KIND_STRING: { - const auto & s = as(*node); - return s.max_length >= 0 && s.min_length > s.max_length ? none() : std::move(node); - } - case COMMON_SCHEMA_KIND_ANY_OF: - return make_any_of(rewrite_all(std::move(as(*node).children))); - case COMMON_SCHEMA_KIND_ALL_OF: - return intersect_all(rewrite_all(std::move(as(*node).children))); - case COMMON_SCHEMA_KIND_ARRAY: { - auto & arr = as(*node); - return make_array(rewrite(std::move(arr.items)), arr.min_items, arr.max_items); - } - case COMMON_SCHEMA_KIND_TUPLE: { - auto & tup = as(*node); - tup.items = rewrite_all(std::move(tup.items)); - for (const auto & item : tup.items) { - if (is(item, COMMON_SCHEMA_KIND_NONE)) { - return none(); - } - } - return node; - } - case COMMON_SCHEMA_KIND_OBJECT: { - auto & obj = as(*node); - for (auto & prop : obj.properties) { - prop.schema = rewrite(std::move(prop.schema)); - } - if (obj.additional_properties) { - obj.additional_properties = rewrite(std::move(obj.additional_properties)); - } - return make_object(std::move(obj.properties), std::move(obj.additional_properties)); - } - } - return node; - } - - static common_schema_ptr make_all_of(std::vector children) { - if (children.size() == 1) { - return std::move(children[0]); - } - auto node = std::make_unique(); - node->children = std::move(children); - return node; - } - - // what is left when two nodes cannot be combined - static common_schema_ptr irreducible(const common_schema & a, const common_schema & b) { - std::vector children; - children.push_back(clone(a)); - children.push_back(clone(b)); - return make_all_of(std::move(children)); - } - - // The intersection of rewritten nodes, an allOf of those that could not be combined. - common_schema_ptr intersect_all(std::vector children) { - std::vector flat; - for (auto & child : children) { - if (is(child, COMMON_SCHEMA_KIND_ALL_OF)) { - for (auto & c : as(*child).children) { - flat.push_back(std::move(c)); - } - } else { - flat.push_back(std::move(child)); - } - } - - std::vector residual; - for (auto & child : flat) { - bool merged = false; - for (auto & r : residual) { - auto cand = intersect(*r, *child); - if (is(cand, COMMON_SCHEMA_KIND_NONE)) { - return none(); - } - if (!is(cand, COMMON_SCHEMA_KIND_ALL_OF)) { - r = std::move(cand); - merged = true; - break; - } - } - if (!merged) { - residual.push_back(std::move(child)); - } - } - if (residual.empty()) { - return std::make_unique(); - } - return make_all_of(std::move(residual)); - } - - static const common_schema_property * find_property(const common_schema_object & obj, const std::string & name) { - for (const auto & prop : obj.properties) { - if (prop.name == name) { - return ∝ - } - } - return nullptr; - } - - // Properties are merged the way json_schema_to_grammar() merges an allOf: a property the other - // side does not list survives, a side that is closed does not shut it out. - common_schema_ptr intersect_object(const common_schema_object & oa, const common_schema_object & ob) { - std::vector properties; - auto add = [&](const common_schema_object & x, const common_schema_object & y, bool skip_shared) { - for (const auto & px : x.properties) { - const auto * py = find_property(y, px.name); - if (py && skip_shared) { - continue; - } - common_schema_property prop; - prop.name = px.name; - prop.required = px.required || (py && py->required); - if (py) { - prop.schema = intersect(*px.schema, *py->schema); - } else if (y.additional_properties) { - prop.schema = intersect(*px.schema, *y.additional_properties); - } else { - prop.schema = clone(*px.schema); - } - properties.push_back(std::move(prop)); - } - }; - add(oa, ob, false); - add(ob, oa, true); - - common_schema_ptr additional; - if (oa.additional_properties && ob.additional_properties) { - additional = intersect(*oa.additional_properties, *ob.additional_properties); - } - return make_object(std::move(properties), std::move(additional)); - } - - // The intersection of two rewritten nodes, an allOf of both where they cannot be combined. - common_schema_ptr intersect(const common_schema & a, const common_schema & b) { - if (is(a, COMMON_SCHEMA_KIND_ANY)) { - return clone(b); - } - if (is(b, COMMON_SCHEMA_KIND_ANY)) { - return clone(a); - } - if (is(a, COMMON_SCHEMA_KIND_NONE) || is(b, COMMON_SCHEMA_KIND_NONE)) { - return none(); - } - if (is(a, COMMON_SCHEMA_KIND_REF) || is(b, COMMON_SCHEMA_KIND_REF)) { - if (is(a, COMMON_SCHEMA_KIND_REF) && is(b, COMMON_SCHEMA_KIND_REF) && as(a).ref == as(b).ref) { - return clone(a); - } - const common_schema * ta = resolve(a); - const common_schema * tb = resolve(b); - if (!ta || !tb) { - return irreducible(a, b); - } - auto key = std::make_pair(ta, tb); - if (!active_.insert(key).second) { - // the same pair is already being intersected further up, a recursive schema - return irreducible(a, b); - } - auto out = intersect(*ta, *tb); - active_.erase(key); - return out; - } - if (is(a, COMMON_SCHEMA_KIND_ANY_OF) || is(b, COMMON_SCHEMA_KIND_ANY_OF)) { - // (a1 | a2) & b = (a1 & b) | (a2 & b) - std::vector alts; - if (is(a, COMMON_SCHEMA_KIND_ANY_OF)) { - for (const auto & child : as(a).children) { - alts.push_back(intersect(*child, b)); - } - } else { - for (const auto & child : as(b).children) { - alts.push_back(intersect(a, *child)); - } - } - return make_any_of(std::move(alts)); - } - if (is(a, COMMON_SCHEMA_KIND_ALL_OF) || is(b, COMMON_SCHEMA_KIND_ALL_OF)) { - std::vector parts; - for (const auto * node : {&a, &b}) { - if (is(*node, COMMON_SCHEMA_KIND_ALL_OF)) { - for (const auto & child : as(*node).children) { - parts.push_back(clone(*child)); - } - } else { - parts.push_back(clone(*node)); - } - } - return intersect_all(std::move(parts)); - } - - std::vector va; - std::vector vb; - bool values_a = get_values(a, va); - bool values_b = get_values(b, vb); - if (values_a && values_b) { - std::vector common; - for (const auto & v : va) { - if (contains(vb, v)) { - common.push_back(v); - } - } - return make_values(std::move(common)); - } - if (values_a || values_b) { - // whether a value fits a schema is not checked - return irreducible(a, b); - } - - if (a.kind() != b.kind()) { - if (is(a, COMMON_SCHEMA_KIND_NUMBER) && is(b, COMMON_SCHEMA_KIND_INTEGER)) { - return clone(b); - } - if (is(a, COMMON_SCHEMA_KIND_INTEGER) && is(b, COMMON_SCHEMA_KIND_NUMBER)) { - return clone(a); - } - if ((is(a, COMMON_SCHEMA_KIND_ARRAY) && is(b, COMMON_SCHEMA_KIND_TUPLE)) || (is(a, COMMON_SCHEMA_KIND_TUPLE) && is(b, COMMON_SCHEMA_KIND_ARRAY))) { - return irreducible(a, b); - } - return none(); - } - switch (a.kind()) { - case COMMON_SCHEMA_KIND_INTEGER: { - const auto & ia = as(a); - const auto & ib = as(b); - auto node = std::make_unique(); - node->minimum = std::max(ia.minimum, ib.minimum); - node->maximum = std::min(ia.maximum, ib.maximum); - return node->minimum > node->maximum ? none() : std::move(node); - } - case COMMON_SCHEMA_KIND_STRING: { - const auto & sa = as(a); - const auto & sb = as(b); - if (!sa.pattern.empty() && !sb.pattern.empty() && sa.pattern != sb.pattern) { - return irreducible(a, b); - } - if (sa.format != COMMON_SCHEMA_FORMAT_NONE && sb.format != COMMON_SCHEMA_FORMAT_NONE && sa.format != sb.format) { - return none(); - } - auto node = std::make_unique(); - node->pattern = sa.pattern.empty() ? sb.pattern : sa.pattern; - node->format = sa.format == COMMON_SCHEMA_FORMAT_NONE ? sb.format : sa.format; - node->min_length = std::max(sa.min_length, sb.min_length); - node->max_length = sa.max_length < 0 ? sb.max_length : sb.max_length < 0 ? sa.max_length : std::min(sa.max_length, sb.max_length); - return node->max_length >= 0 && node->min_length > node->max_length ? none() : std::move(node); - } - case COMMON_SCHEMA_KIND_ARRAY: { - const auto & aa = as(a); - const auto & ab = as(b); - int min_items = std::max(aa.min_items, ab.min_items); - int max_items = aa.max_items < 0 ? ab.max_items : ab.max_items < 0 ? aa.max_items : std::min(aa.max_items, ab.max_items); - return make_array(intersect(*aa.items, *ab.items), min_items, max_items); - } - case COMMON_SCHEMA_KIND_TUPLE: { - const auto & ta = as(a); - const auto & tb = as(b); - if (ta.items.size() != tb.items.size()) { - return none(); - } - auto node = std::make_unique(); - for (size_t i = 0; i < ta.items.size(); i++) { - auto item = intersect(*ta.items[i], *tb.items[i]); - if (is(item, COMMON_SCHEMA_KIND_NONE)) { - return none(); - } - node->items.push_back(std::move(item)); - } - return node; - } - case COMMON_SCHEMA_KIND_OBJECT: - return intersect_object(as(a), as(b)); - default: - // null, boolean and number have no fields to differ in - return clone(a); - } - } - - // Moves the refs the node reaches into live, and points each ref node at its target there. - void link(common_schema & node, std::map & live) { - switch (node.kind()) { - case COMMON_SCHEMA_KIND_REF: { - auto & ref = as(node); - auto it = live.find(ref.ref); - if (it == live.end()) { - it = live.emplace(ref.ref, std::move(doc_.refs.at(ref.ref))).first; - link(*it->second, live); - } - ref.target = it->second.get(); - return; - } - case COMMON_SCHEMA_KIND_ANY_OF: - for (auto & child : as(node).children) { - link(*child, live); - } - return; - case COMMON_SCHEMA_KIND_ALL_OF: - for (auto & child : as(node).children) { - link(*child, live); - } - return; - case COMMON_SCHEMA_KIND_ARRAY: - link(*as(node).items, live); - return; - case COMMON_SCHEMA_KIND_TUPLE: - for (auto & item : as(node).items) { - link(*item, live); - } - return; - case COMMON_SCHEMA_KIND_OBJECT: { - auto & obj = as(node); - for (auto & prop : obj.properties) { - link(*prop.schema, live); - } - if (obj.additional_properties) { - link(*obj.additional_properties, live); - } - return; - } - default: - return; - } - } - - public: - explicit common_schema_optimizer(common_schema_document & doc) : doc_(doc) {} - - void run() { - // a $ref whose target became none is none too, which the next pass can then prune in its parent - do { - changed_ = false; - for (auto & entry : doc_.refs) { - entry.second = rewrite(std::move(entry.second)); - } - doc_.root = rewrite(std::move(doc_.root)); - } while (changed_); - - // only the refs the root still reaches are kept, and every ref node gets its new target - std::map live; - link(*doc_.root, live); - doc_.refs = std::move(live); - } -}; - -void common_schema_optimize(common_schema_document & doc) { - common_schema_optimizer(doc).run(); -} - static common_schema_kind kind_of(const common_json & value) { if (value.is_null()) { return COMMON_SCHEMA_KIND_NULL; @@ -972,8 +374,6 @@ static common_schema_kinds resolve_kinds(const common_schema & s, std::unordered switch (s.kind()) { case COMMON_SCHEMA_KIND_ANY: return common_schema_kinds::all(); - case COMMON_SCHEMA_KIND_NONE: - return {}; case COMMON_SCHEMA_KIND_NUMBER: return { COMMON_SCHEMA_KIND_NUMBER, COMMON_SCHEMA_KIND_INTEGER }; case COMMON_SCHEMA_KIND_TUPLE: diff --git a/common/json-schema.h b/common/json-schema.h index 5cae78f2f9..3151e90280 100644 --- a/common/json-schema.h +++ b/common/json-schema.h @@ -14,7 +14,6 @@ enum common_schema_kind { COMMON_SCHEMA_KIND_ANY, - COMMON_SCHEMA_KIND_NONE, COMMON_SCHEMA_KIND_REF, COMMON_SCHEMA_KIND_ANY_OF, COMMON_SCHEMA_KIND_ALL_OF, @@ -51,11 +50,6 @@ struct common_schema_any : common_schema { common_schema_kind kind() const override { return COMMON_SCHEMA_KIND_ANY; } }; -// Matches no value: what common_schema_optimize() leaves where an intersection turned out empty -struct common_schema_none : common_schema { - common_schema_kind kind() const override { return COMMON_SCHEMA_KIND_NONE; } -}; - // {"$ref": "#/..."}, only references into the same document are supported struct common_schema_ref : common_schema { std::string ref; @@ -165,12 +159,6 @@ common_schema_document common_schema_parse(const common_json & schema); // doc is unchanged when the schema is rejected. common_schema_ptr common_schema_parse(const common_json & schema, common_schema_document & doc); -// Rewrites a document in place into an equivalent one with less redundancy: allOf becomes the intersection -// of its children, nested anyOf are flattened, branches that can match nothing are pruned, up to a -// common_schema_none root, and $refs nothing reaches anymore are dropped. -// An allOf stays where the children cannot be combined, e.g. two different patterns or a const against a type. -void common_schema_optimize(common_schema_document & doc); - // A set of the kinds of value a schema may match: only the value kinds NULL to OBJECT occur, a tuple counts as an array class common_schema_kinds { uint32_t mask_ = 0; diff --git a/common/parsers/functionary-v3-2.cpp b/common/parsers/functionary-v3-2.cpp index 349b8065ac..d2c9aec61d 100644 --- a/common/parsers/functionary-v3-2.cpp +++ b/common/parsers/functionary-v3-2.cpp @@ -45,7 +45,7 @@ common_chat_params common_chat_params_init_functionary_v3_2(const common_chat_te foreach_function(inputs.tools, [&](const json & tool) { const auto & function = tool.at("function"); std::string name = function.at("name"); - const auto & schema = function.at("parameters"); + auto schema = common_chat_tool_parameters(function); // Tool format: >>>function_name\n{json_args} auto tool_parser = p.tool( diff --git a/common/parsers/gigachat-v3.cpp b/common/parsers/gigachat-v3.cpp index 41da5554ac..086bac6366 100644 --- a/common/parsers/gigachat-v3.cpp +++ b/common/parsers/gigachat-v3.cpp @@ -33,7 +33,7 @@ common_chat_params common_chat_params_init_gigachat_v3( for (const auto & tool : inputs.tools) { const auto & function = tool.at("function"); std::string name = function.at("name"); - const auto & schema = function.at("parameters"); + auto schema = common_chat_tool_parameters(function); auto tool_name = p.json_member("name", "\"" + p.tool_name(p.literal(name)) + "\""); auto tool_args = p.json_member("arguments", p.tool_args(p.schema(p.json(), "tool-" + name + "-schema", schema))); diff --git a/common/parsers/gpt-oss.cpp b/common/parsers/gpt-oss.cpp index d7dbfbfb57..c1148344f7 100644 --- a/common/parsers/gpt-oss.cpp +++ b/common/parsers/gpt-oss.cpp @@ -109,7 +109,7 @@ common_chat_params common_chat_params_init_gpt_oss(const common_chat_template & foreach_function(inputs.tools, [&](const json & tool) { const auto & function = tool.at("function"); std::string name = function.at("name"); - const auto & params = function.at("parameters"); + auto params = common_chat_tool_parameters(function); auto func_name = p.literal(" to=functions.") + p.tool_name(p.literal(name)); auto constraint = p.optional(p.space() + p.optional(p.literal("<|constrain|>")) + constrain_type); diff --git a/common/parsers/kimi-k2.cpp b/common/parsers/kimi-k2.cpp index 57f6bfdcb6..ebc7e5a4b0 100644 --- a/common/parsers/kimi-k2.cpp +++ b/common/parsers/kimi-k2.cpp @@ -82,7 +82,7 @@ common_chat_params common_chat_params_init_kimi_k2(const common_chat_template & foreach_function(inputs.tools, [&](const json & tool) { const auto & function = tool.at("function"); std::string name = function.at("name"); - const auto & schema = function.at("parameters"); + auto schema = common_chat_tool_parameters(function); // Match: functions.: // Capture the full call id (functions.:) using tool_id tag diff --git a/common/parsers/ministral3.cpp b/common/parsers/ministral3.cpp index 075f14dbc1..7b6b50d729 100644 --- a/common/parsers/ministral3.cpp +++ b/common/parsers/ministral3.cpp @@ -89,7 +89,7 @@ common_chat_params common_chat_params_init_ministral_3(const common_chat_templat foreach_function(inputs.tools, [&](const json & tool) { const auto & function = tool.at("function"); std::string name = function.at("name"); - const auto & schema = function.at("parameters"); + auto schema = common_chat_tool_parameters(function); tool_choice |= p.rule("tool-" + name, p.tool_open(p.tool_name(p.literal(name)) + "[ARGS]") + diff --git a/tests/test-chat.cpp b/tests/test-chat.cpp index f27c91e4d4..1aef83f430 100644 --- a/tests/test-chat.cpp +++ b/tests/test-chat.cpp @@ -472,6 +472,12 @@ static common_chat_tool empty_args_tool_no_properties{ })", }; +static common_chat_tool empty_args_tool_no_schema{ + /* .name = */ "empty_args_no_schema", + /* .description = */ "A tool that takes no arguments and has no parameters schema", + /* .parameters = */ "{}", +}; + static common_chat_tool python_tool{ /* .name = */ "python", /* .description = */ "an ipython interpreter", @@ -5071,6 +5077,13 @@ static void test_template_output_peg_parsers(bool detailed_debug) { .expect(simple_assist_msg("", "", "empty_args", "{}")) .run(); + // Tool call with no parameters schema, {} means no arguments + tst.test("\n{\"name\": \"empty_args_no_schema\", \"arguments\": {}}") + .enable_thinking(false) + .tools({ empty_args_tool_no_schema }) + .expect(simple_assist_msg("", "", "empty_args_no_schema", "{}")) + .run(); + // fake tool call marker in reasoning tst.test( "Let me think about \n{\"name\": \"special_function\", \"arguments\": {\"arg1\": 2}} hmm\n\n\n" diff --git a/tests/test-json-schema.cpp b/tests/test-json-schema.cpp index bf8edd2e73..3a01ff0fc7 100644 --- a/tests/test-json-schema.cpp +++ b/tests/test-json-schema.cpp @@ -49,7 +49,6 @@ static std::string dump_range(int64_t min, int64_t min_def, int64_t max, int64_t static std::string dump(const common_schema & node) { switch (node.kind()) { case COMMON_SCHEMA_KIND_ANY: return "any"; - case COMMON_SCHEMA_KIND_NONE: return "none"; case COMMON_SCHEMA_KIND_NULL: return "null"; case COMMON_SCHEMA_KIND_BOOLEAN: return "boolean"; case COMMON_SCHEMA_KIND_NUMBER: return "number"; @@ -100,12 +99,6 @@ static std::string dump(const common_schema & node) { return "?"; } -static std::string optimize(const std::string & schema) { - auto doc = parse(schema); - common_schema_optimize(doc); - return dump(*doc.root); -} - static void assert_error(testing & t, const std::string & schema, const std::string & needle) { try { parse(schema); @@ -706,296 +699,6 @@ static void test_errors(testing & t) { }); } -static void test_optimize_any_of(testing & t) { - t.test("nested anyOf are flattened", [](testing & t) { - t.assert_equal("anyOf(string, null, boolean)", optimize(R"({"anyOf": [{"anyOf": [{"type": "string"}, {"type": "null"}]}, {"type": "boolean"}]})")); - }); - - t.test("one alternative left unwraps", [](testing & t) { - t.assert_equal("null", optimize(R"({"oneOf": [{"type": "null"}]})")); - }); - - t.test("any absorbs the rest", [](testing & t) { - t.assert_equal("any", optimize(R"({"anyOf": [{"type": "string"}, {}, {"type": "null"}]})")); - }); - - t.test("an alternative that matches nothing is pruned", [](testing & t) { - t.assert_equal("string", optimize(R"({"anyOf": [{"type": "integer", "minimum": 5, "maximum": 1}, {"type": "string"}]})")); - t.assert_equal("none", optimize(R"({"anyOf": [{"type": "integer", "minimum": 5, "maximum": 1}]})")); - }); -} - -static void test_optimize_all_of(testing & t) { - t.test("objects merge their properties", [](testing & t) { - t.assert_equal("object{a: string, b?: integer}", optimize(R"({"allOf": [ - {"properties": {"a": {"type": "string"}}, "required": ["a"]}, - {"properties": {"b": {"type": "integer"}}} - ]})")); - }); - - t.test("a shared property is intersected and required by either side", [](testing & t) { - t.assert_equal("object{a: integer[1..5]}", optimize(R"({"allOf": [ - {"properties": {"a": {"type": "integer", "minimum": 1}}, "required": ["a"]}, - {"properties": {"a": {"type": "integer", "maximum": 5}}} - ]})")); - }); - - t.test("additionalProperties constrains the other side's properties", [](testing & t) { - t.assert_equal("object{a?: integer[0..]}", optimize(R"({"allOf": [ - {"properties": {"a": {"type": "integer"}}}, - {"additionalProperties": {"type": "integer", "minimum": 0}} - ]})")); - t.assert_equal("object{a?: integer, *: integer}", optimize(R"({"allOf": [ - {"properties": {"a": {"type": "number"}}, "additionalProperties": true}, - {"additionalProperties": {"type": "integer"}} - ]})")); - t.assert_equal("object{*: integer}", optimize(R"({"allOf": [ - {"properties": {"a": {"type": "string"}}, "additionalProperties": true}, - {"additionalProperties": {"type": "integer"}} - ]})")); - }); - - t.test("a shared property that matches nothing", [](testing & t) { - t.assert_equal("object{b?: any}", optimize(R"({"allOf": [ - {"properties": {"a": {"type": "string"}, "b": {}}}, - {"properties": {"a": {"type": "integer"}}} - ]})")); - t.assert_equal("none", optimize(R"({"allOf": [ - {"properties": {"a": {"type": "string"}}, "required": ["a"]}, - {"properties": {"a": {"type": "integer"}}} - ]})")); - }); - - t.test("a ref is inlined and dropped", [](testing & t) { - auto doc = parse(R"({ - "allOf": [{"$ref": "#/$defs/base"}, {"properties": {"b": {"type": "string"}}}], - "$defs": {"base": {"properties": {"a": {"type": "string"}}, "required": ["a"]}} - })"); - common_schema_optimize(doc); - t.assert_equal("root", "object{a: string, b?: string}", dump(*doc.root)); - t.assert_true("no refs", doc.refs.empty()); - }); - - t.test("enums and consts intersect", [](testing & t) { - t.assert_equal("const(\"b\")", optimize(R"({"type": "string", "allOf": [{"enum": ["a", "b"]}, {"enum": ["b", "c"]}]})")); - t.assert_equal("enum(1, 2)", optimize(R"({"allOf": [{"enum": [1, "x", 2]}, {"enum": [2, 1, 3]}]})")); - t.assert_equal("const(1)", optimize(R"({"allOf": [{"const": 1}, {"enum": [2, 1]}]})")); - t.assert_equal("none", optimize(R"({"allOf": [{"enum": ["a"]}, {"enum": ["b"]}]})")); - t.assert_equal("none", optimize(R"({"allOf": [{"const": 5}, {"const": 6}]})")); - }); - - t.test("a value against a type stays an allOf", [](testing & t) { - t.assert_equal("allOf(const(5), integer)", optimize(R"({"allOf": [{"const": 5}, {"type": "integer"}]})")); - }); - - t.test("integer bounds", [](testing & t) { - t.assert_equal("integer[1..10]", optimize(R"({"allOf": [{"type": "integer", "minimum": 1}, {"type": "integer", "maximum": 10}]})")); - t.assert_equal("integer[3..5]", optimize(R"({"allOf": [{"type": "integer", "minimum": 1, "maximum": 5}, {"type": "integer", "minimum": 3, "maximum": 10}]})")); - t.assert_equal("none", optimize(R"({"allOf": [{"type": "integer", "maximum": 1}, {"type": "integer", "minimum": 2}]})")); - t.assert_equal("integer", optimize(R"({"allOf": [{"type": "number"}, {"type": "integer"}]})")); - }); - - t.test("strings", [](testing & t) { - t.assert_equal("string:date[2..5]", optimize(R"({"type": "string", "allOf": [{"minLength": 2, "type": "string"}, {"type": "string", "maxLength": 5, "format": "date"}]})")); - t.assert_equal("string/^a/[1..3]", optimize(R"({"allOf": [{"pattern": "^a", "maxLength": 3}, {"pattern": "^a", "minLength": 1}]})")); - t.assert_equal("allOf(string/^a/, string/b$/)", optimize(R"({"allOf": [{"pattern": "^a"}, {"pattern": "b$"}]})")); - t.assert_equal("none", optimize(R"({"allOf": [{"format": "date"}, {"format": "uuid"}]})")); - t.assert_equal("none", optimize(R"({"allOf": [{"type": "string", "minLength": 5}, {"type": "string", "maxLength": 2}]})")); - }); - - t.test("kinds that cannot both hold", [](testing & t) { - t.assert_equal("none", optimize(R"({"allOf": [{"type": "string"}, {"type": "integer"}]})")); - t.assert_equal("none", optimize(R"({"allOf": [{"type": "object"}, {"items": {}}]})")); - }); - - t.test("any and equal children fold away", [](testing & t) { - t.assert_equal("boolean", optimize(R"({"allOf": [{}, {"type": "boolean"}, {"type": "boolean"}]})")); - t.assert_equal("any", optimize(R"({"allOf": [{}, {}]})")); - }); - - t.test("arrays and tuples", [](testing & t) { - t.assert_equal("array(integer[0..])[1..3]", optimize(R"({"allOf": [ - {"items": {"type": "integer"}, "minItems": 1}, - {"items": {"type": "integer", "minimum": 0}, "maxItems": 3} - ]})")); - t.assert_equal("tuple()", optimize(R"({"allOf": [{"items": {"type": "string"}}, {"items": {"type": "integer"}}]})")); - t.assert_equal("tuple(integer[1..], null)", optimize(R"({"allOf": [{"prefixItems": [{"type": "integer"}, {}]}, {"prefixItems": [{"minimum": 1, "type": "integer"}, {"type": "null"}]}]})")); - t.assert_equal("none", optimize(R"({"allOf": [{"prefixItems": [{}]}, {"prefixItems": [{}, {}]}]})")); - t.assert_equal("allOf(array(integer), tuple(integer))", optimize(R"({"allOf": [{"items": {"type": "integer"}}, {"prefixItems": [{"type": "integer"}]}]})")); - }); - - t.test("distributes over anyOf", [](testing & t) { - t.assert_equal("integer[0..]", optimize(R"({"allOf": [ - {"anyOf": [{"type": "string"}, {"type": "integer"}]}, - {"type": "integer", "minimum": 0} - ]})")); - t.assert_equal("anyOf(object{a?: string, c: integer}, object{b?: string, c: integer})", optimize(R"({"allOf": [ - {"anyOf": [{"properties": {"a": {"type": "string"}}}, {"properties": {"b": {"type": "string"}}}]}, - {"properties": {"c": {"type": "integer"}}, "required": ["c"]} - ]})")); - t.assert_equal("anyOf(integer[..3], integer[5..])", optimize(R"({"allOf": [ - {"anyOf": [{"type": "integer", "maximum": 3}, {"type": "integer", "minimum": 5}]}, - {"anyOf": [{"type": "integer"}, {"type": "string"}]} - ]})")); - }); - - t.test("nested allOf are flattened", [](testing & t) { - t.assert_equal("integer[1..3]", optimize(R"({"allOf": [{"allOf": [{"type": "integer", "minimum": 1}]}, {"type": "integer", "maximum": 3}]})")); - }); - - t.test("what cannot combine keeps the rest merged", [](testing & t) { - t.assert_equal("allOf(string/^a/[..5], string/b$/)", optimize(R"({"allOf": [{"pattern": "^a"}, {"pattern": "b$"}, {"type": "string", "maxLength": 5}]})")); - }); - - t.test("recursive refs do not loop", [](testing & t) { - auto doc = parse(R"({ - "allOf": [{"$ref": "#/$defs/a"}, {"$ref": "#/$defs/b"}], - "$defs": { - "a": {"properties": {"next": {"$ref": "#/$defs/a"}}}, - "b": {"properties": {"next": {"$ref": "#/$defs/b"}}} - } - })"); - common_schema_optimize(doc); - t.assert_equal("root", "object{next?: allOf(ref(#/$defs/a), ref(#/$defs/b))}", dump(*doc.root)); - t.assert_equal("refs", (size_t) 2, doc.refs.size()); - }); -} - -static void test_optimize_prune(testing & t) { - t.test("bounds that leave nothing", [](testing & t) { - t.assert_equal("none", optimize(R"({"type": "integer", "minimum": 5, "maximum": 1})")); - t.assert_equal("none", optimize(R"({"type": "string", "minLength": 5, "maxLength": 1})")); - t.assert_equal("none", optimize(R"({"type": "array", "minItems": 5, "maxItems": 1})")); - t.assert_equal("integer[1..1]", optimize(R"({"type": "integer", "minimum": 1, "maximum": 1})")); - }); - - t.test("an optional property that can never be present is dropped", [](testing & t) { - t.assert_equal("object{b?: any}", optimize(R"({"properties": {"a": {"type": "integer", "minimum": 5, "maximum": 1}, "b": {}}})")); - }); - - t.test("a required property that can never be present empties the object", [](testing & t) { - t.assert_equal("none", optimize(R"({"properties": {"a": {"type": "integer", "minimum": 5, "maximum": 1}}, "required": ["a"]})")); - t.assert_equal("none", optimize(R"({"properties": {"o": {"properties": {"a": {"type": "integer", "minimum": 5, "maximum": 1}}, "required": ["a"]}}, "required": ["o"]})")); - }); - - t.test("additionalProperties that can never be present closes the object", [](testing & t) { - t.assert_equal("object{a?: string}", optimize(R"({"properties": {"a": {"type": "string"}}, "additionalProperties": {"type": "integer", "minimum": 5, "maximum": 1}})")); - }); - - t.test("items that can never be present leave the empty array", [](testing & t) { - t.assert_equal("tuple()", optimize(R"({"items": {"type": "integer", "minimum": 5, "maximum": 1}})")); - t.assert_equal("tuple()", optimize(R"({"type": "array", "maxItems": 0})")); - t.assert_equal("none", optimize(R"({"items": {"type": "integer", "minimum": 5, "maximum": 1}, "minItems": 1})")); - }); - - t.test("a tuple item that can never be present", [](testing & t) { - t.assert_equal("none", optimize(R"({"prefixItems": [{"type": "string"}, {"type": "integer", "minimum": 5, "maximum": 1}]})")); - }); - - t.test("enum values dedupe", [](testing & t) { - t.assert_equal("enum(\"a\", \"b\")", optimize(R"({"enum": ["a", "a", "b"]})")); - t.assert_equal("const(\"a\")", optimize(R"({"enum": ["a", "a"]})")); - }); - - t.test("nested pruning reaches the root", [](testing & t) { - t.assert_equal("none", optimize(R"({"anyOf": [ - {"properties": {"a": {"allOf": [{"type": "string"}, {"type": "null"}]}}, "required": ["a"]}, - {"prefixItems": [{"allOf": [{"const": 1}, {"const": 2}]}]} - ]})")); - }); - - t.test("what is already minimal is left alone", [](testing & t) { - auto doc = parse(R"({ - "properties": { - "name": {"type": "string", "minLength": 1}, - "tags": {"type": "array", "items": {"enum": ["a", "b"]}, "maxItems": 3}, - "kind": {"anyOf": [{"type": "null"}, {"$ref": "#/$defs/kind"}]} - }, - "required": ["name"], - "$defs": {"kind": {"properties": {"id": {"type": "integer", "minimum": 0}}, "required": ["id"]}} - })"); - std::string before = dump(*doc.root); - common_schema_optimize(doc); - t.assert_equal("root", before, dump(*doc.root)); - t.assert_equal("root", "object{name: string[1..], tags?: array(enum(\"a\", \"b\"))[..3], kind?: anyOf(null, ref(#/$defs/kind))}", dump(*doc.root)); - t.assert_equal("refs", (size_t) 1, doc.refs.size()); - }); -} - -static void test_optimize_ref(testing & t) { - t.test("a ref to nothing is nothing", [](testing & t) { - auto doc = parse(R"({"$ref": "#/$defs/t", "$defs": {"t": {"type": "integer", "minimum": 5, "maximum": 1}}})"); - common_schema_optimize(doc); - t.assert_equal("root", "none", dump(*doc.root)); - t.assert_true("no refs", doc.refs.empty()); - }); - - t.test("a branch through a ref to nothing is pruned", [](testing & t) { - auto doc = parse(R"({ - "anyOf": [{"properties": {"x": {"$ref": "#/$defs/t"}}, "required": ["x"]}, {"type": "string"}], - "$defs": {"t": {"allOf": [{"type": "string"}, {"type": "null"}]}} - })"); - common_schema_optimize(doc); - t.assert_equal("root", "string", dump(*doc.root)); - t.assert_true("no refs", doc.refs.empty()); - }); - - t.test("a chain of refs to nothing", [](testing & t) { - auto doc = parse(R"({ - "properties": {"a": {"$ref": "#/$defs/a"}, "b": {}}, - "$defs": {"a": {"$ref": "#/$defs/b"}, "b": {"$ref": "#/$defs/c"}, "c": {"type": "integer", "minimum": 5, "maximum": 1}} - })"); - common_schema_optimize(doc); - t.assert_equal("root", "object{b?: any}", dump(*doc.root)); - t.assert_true("no refs", doc.refs.empty()); - }); - - t.test("reachable refs are kept and relinked", [](testing & t) { - auto doc = parse(R"({ - "properties": {"a": {"$ref": "#/$defs/t"}, "b": {"$ref": "#/$defs/t"}, "c": {"$ref": "#/$defs/u"}}, - "$defs": {"t": {"anyOf": [{"type": "string"}]}, "u": {"type": "null"}, "unused": {"type": "boolean"}} - })"); - common_schema_optimize(doc); - t.assert_equal("root", "object{a?: ref(#/$defs/t), b?: ref(#/$defs/t), c?: ref(#/$defs/u)}", dump(*doc.root)); - t.assert_equal("refs", (size_t) 2, doc.refs.size()); - t.assert_equal("t", "string", dump(*doc.refs.at("#/$defs/t"))); - const auto & o = root(t, doc); - for (const auto & prop : o.properties) { - const auto & r = as(t, prop.schema.get(), prop.name); - t.assert_true(prop.name + " target", r.target != nullptr && r.target == doc.refs.at(r.ref).get()); - } - }); - - t.test("a recursive schema survives", [](testing & t) { - auto doc = parse(R"({ - "$ref": "#/$defs/node", - "$defs": { - "node": { - "properties": { - "value": {"type": "number"}, - "next": {"anyOf": [{"$ref": "#/$defs/node"}, {"type": "null"}]} - }, - "required": ["value"] - } - } - })"); - common_schema_optimize(doc); - const auto & r = root(t, doc); - t.assert_equal("node", "object{value: number, next?: anyOf(ref(#/$defs/node), null)}", dump(*r.target)); - t.assert_true("target", r.target == doc.refs.at("#/$defs/node").get()); - const auto & node = as(t, r.target, "node"); - const auto & next = as(t, node.properties[1].schema.get(), "next"); - t.assert_true("cycle", as(t, next.children[0].get(), "next[0]").target == r.target); - }); - - t.test("a ref intersected with any or itself stays a ref", [](testing & t) { - auto doc = parse(R"({"allOf": [{}, {"$ref": "#/$defs/t"}, {"$ref": "#/$defs/t"}], "$defs": {"t": {"type": "boolean"}}})"); - common_schema_optimize(doc); - t.assert_equal("root", "ref(#/$defs/t)", dump(*doc.root)); - t.assert_true("target", root(t, doc).target == doc.refs.at("#/$defs/t").get()); - }); -} - int main(int argc, char * argv[]) { testing t(std::cout); if (argc >= 2) { @@ -1019,10 +722,6 @@ int main(int argc, char * argv[]) { t.test("all_of", test_all_of); t.test("ref", test_ref); t.test("errors", test_errors); - t.test("optimize any_of", test_optimize_any_of); - t.test("optimize all_of", test_optimize_all_of); - t.test("optimize prune", test_optimize_prune); - t.test("optimize ref", test_optimize_ref); return t.summary(); } diff --git a/tools/cli/README.md b/tools/cli/README.md index 77a5e6fe32..ea1f7aaf8b 100644 --- a/tools/cli/README.md +++ b/tools/cli/README.md @@ -133,8 +133,8 @@ | `-l, --logit-bias TOKEN_ID(+/-)BIAS` | modifies the likelihood of token appearing in the completion,
i.e. `--logit-bias 15043+1` to increase likelihood of token ' Hello',
or `--logit-bias 15043-1` to decrease likelihood of token ' Hello' | | `--grammar GRAMMAR` | BNF-like grammar to constrain generations (see samples in grammars/ dir) | | `--grammar-file FNAME` | file to read grammar from | -| `-j, --json-schema SCHEMA` | JSON schema to constrain generations (https://json-schema.org/), e.g. `{}` for any JSON object
For schemas w/ external $refs, use --grammar + example/json_schema_to_grammar.py instead | -| `-jf, --json-schema-file FILE` | File containing a JSON schema to constrain generations (https://json-schema.org/), e.g. `{}` for any JSON object
For schemas w/ external $refs, use --grammar + example/json_schema_to_grammar.py instead | +| `-j, --json-schema SCHEMA` | JSON schema to constrain generations (https://json-schema.org/), e.g. `{"type": "object"}` for any JSON object | +| `-jf, --json-schema-file FILE` | File containing a JSON schema to constrain generations (https://json-schema.org/), e.g. `{"type": "object"}` for any JSON object | | `-bs, --backend-sampling` | enable backend sampling (experimental) (default: disabled)
(env: LLAMA_ARG_BACKEND_SAMPLING) | diff --git a/tools/completion/README.md b/tools/completion/README.md index 08485a95f5..c9a4cccfc2 100644 --- a/tools/completion/README.md +++ b/tools/completion/README.md @@ -216,8 +216,8 @@ llama-completion.exe -m models\gemma-1.1-7b-it.Q4_K_M.gguf --ignore-eos -n -1 | `-l, --logit-bias TOKEN_ID(+/-)BIAS` | modifies the likelihood of token appearing in the completion,
i.e. `--logit-bias 15043+1` to increase likelihood of token ' Hello',
or `--logit-bias 15043-1` to decrease likelihood of token ' Hello' | | `--grammar GRAMMAR` | BNF-like grammar to constrain generations (see samples in grammars/ dir) | | `--grammar-file FNAME` | file to read grammar from | -| `-j, --json-schema SCHEMA` | JSON schema to constrain generations (https://json-schema.org/), e.g. `{}` for any JSON object
For schemas w/ external $refs, use --grammar + example/json_schema_to_grammar.py instead | -| `-jf, --json-schema-file FILE` | File containing a JSON schema to constrain generations (https://json-schema.org/), e.g. `{}` for any JSON object
For schemas w/ external $refs, use --grammar + example/json_schema_to_grammar.py instead | +| `-j, --json-schema SCHEMA` | JSON schema to constrain generations (https://json-schema.org/), e.g. `{"type": "object"}` for any JSON object | +| `-jf, --json-schema-file FILE` | File containing a JSON schema to constrain generations (https://json-schema.org/), e.g. `{"type": "object"}` for any JSON object | | `-bs, --backend-sampling` | enable backend sampling (experimental) (default: disabled)
(env: LLAMA_ARG_BACKEND_SAMPLING) | @@ -556,7 +556,7 @@ These options help improve the performance and memory usage of the LLaMA models. - `--grammar GRAMMAR`, `--grammar-file FILE`: Specify a grammar (defined inline or in a file) to constrain model output to a specific format. For example, you could force the model to output JSON or to speak only in emojis. See the [GBNF guide](../../grammars/README.md) for details on the syntax. -- `--json-schema SCHEMA`: Specify a [JSON schema](https://json-schema.org/) to constrain model output to (e.g. `{}` for any JSON object, or `{"items": {"type": "string", "minLength": 10, "maxLength": 100}, "minItems": 10}` for a JSON array of strings with size constraints). If a schema uses external `$ref`s, you should use `--grammar "$( python examples/json_schema_to_grammar.py myschema.json )"` instead. +- `--json-schema SCHEMA`: Specify a [JSON schema](https://json-schema.org/) to constrain model output to (e.g. `{"type": "object"}` for any JSON object, or `{"items": {"type": "string", "minLength": 10, "maxLength": 100}, "minItems": 10}` for a JSON array of strings with size constraints). ### Quantization diff --git a/tools/server/README.md b/tools/server/README.md index 1909076328..ef90334048 100644 --- a/tools/server/README.md +++ b/tools/server/README.md @@ -150,8 +150,8 @@ For the full list of features, please refer to [server's changelog](https://gith | `-l, --logit-bias TOKEN_ID(+/-)BIAS` | modifies the likelihood of token appearing in the completion,
i.e. `--logit-bias 15043+1` to increase likelihood of token ' Hello',
or `--logit-bias 15043-1` to decrease likelihood of token ' Hello' | | `--grammar GRAMMAR` | BNF-like grammar to constrain generations (see samples in grammars/ dir) | | `--grammar-file FNAME` | file to read grammar from | -| `-j, --json-schema SCHEMA` | JSON schema to constrain generations (https://json-schema.org/), e.g. `{}` for any JSON object
For schemas w/ external $refs, use --grammar + example/json_schema_to_grammar.py instead | -| `-jf, --json-schema-file FILE` | File containing a JSON schema to constrain generations (https://json-schema.org/), e.g. `{}` for any JSON object
For schemas w/ external $refs, use --grammar + example/json_schema_to_grammar.py instead | +| `-j, --json-schema SCHEMA` | JSON schema to constrain generations (https://json-schema.org/), e.g. `{"type": "object"}` for any JSON object | +| `-jf, --json-schema-file FILE` | File containing a JSON schema to constrain generations (https://json-schema.org/), e.g. `{"type": "object"}` for any JSON object | | `-bs, --backend-sampling` | enable backend sampling (experimental) (default: disabled)
(env: LLAMA_ARG_BACKEND_SAMPLING) |