Compare commits

..
1 Commits
Author SHA1 Message Date
weaselbot f0a338f164 make WeaselJson_OVERFLOW a terminal state like WeaselJson_REJECT
CI / build (-DCMAKE_C_COMPILER=clang -DCMAKE_CXX_COMPILER=clang++, clang-arm64, ubuntu-latest-arm64, true) (pull_request) Successful in 53s
CI / pre-commit (pull_request) Successful in 52s
CI / build (-DCMAKE_C_COMPILER=gcc -DCMAKE_CXX_COMPILER=g++, gcc-arm64, ubuntu-latest-arm64, false) (pull_request) Successful in 1m2s
CI / build (-DCMAKE_C_COMPILER=clang -DCMAKE_CXX_COMPILER=clang++, clang-amd64, ubuntu-latest-amd64, true) (pull_request) Successful in 1m33s
CI / build (-DCMAKE_C_COMPILER=gcc -DCMAKE_CXX_COMPILER=g++, gcc-amd64, ubuntu-latest-amd64, false) (pull_request) Successful in 1m31s
When a parse step returned WeaselJson_OVERFLOW, the pushdown stack was
left corrupted (frames had been popped before the failing push), but the
overflow status was not made sticky the way WeaselJson_REJECT is. A
subsequent end-of-data call WeaselJsonParser_parse(parser, nullptr, 0)
could then dispatch through the corrupted stack and return WeaselJson_OK,
accepting an incomplete, invalid too-deeply-nested document as valid JSON.

Add an `overflowed` flag, set it whenever a continuation returns
WeaselJson_OVERFLOW, and short-circuit parse() to return
WeaselJson_OVERFLOW on every subsequent call. reset() clears the flag so a
reused parser can accept a different document. This mirrors the existing
stickiness handling for WeaselJson_REJECT.

Closes #52
2026-07-13 11:45:58 -04:00
10 changed files with 37 additions and 90 deletions
+2 -3
View File
@@ -60,9 +60,8 @@ into the result, so it is non-movable.
## Not supported (rejected at generation time, no fallback) ## Not supported (rejected at generation time, no fallback)
`oneOf` / `anyOf` / `allOf` / `not` / `if`-`then`-`else`, `oneOf` / `anyOf` / `allOf` / `not` / `if`-`then`-`else`,
`patternProperties`, `additionalProperties` absent or set to `true`, `patternProperties`, `additionalProperties` with a schema (typed map),
`additionalProperties` with a schema (typed map), `prefixItems` (tuples), `additionalProperties: true`, `prefixItems` (tuples), `const`,
`const`,
`dependentSchemas`/`dependentRequired`, union `type` lists other than `dependentSchemas`/`dependentRequired`, union `type` lists other than
`["T", "null"]`, non-string enums, and remote (`$ref` to other documents). `["T", "null"]`, non-string enums, and remote (`$ref` to other documents).
-1
View File
@@ -56,7 +56,6 @@
"type": "array", "type": "array",
"items": { "items": {
"type": "object", "type": "object",
"additionalProperties": false,
"required": [ "required": [
"name" "name"
], ],
+12 -54
View File
@@ -32,7 +32,6 @@ class SchemagenEnumTest(unittest.TestCase):
def test_colliding_enum_values_deduplicate(self): def test_colliding_enum_values_deduplicate(self):
schema = { schema = {
"type": "object", "type": "object",
"additionalProperties": False,
"properties": {"role": {"enum": ["foo-bar", "foo_bar"]}}, "properties": {"role": {"enum": ["foo-bar", "foo_bar"]}},
} }
rc, stdout, stderr = self.run_schemagen(schema) rc, stdout, stderr = self.run_schemagen(schema)
@@ -46,7 +45,6 @@ class SchemagenEnumTest(unittest.TestCase):
def test_distinct_enum_values_generate(self): def test_distinct_enum_values_generate(self):
schema = { schema = {
"type": "object", "type": "object",
"additionalProperties": False,
"properties": {"role": {"enum": ["admin", "user", "guest"]}}, "properties": {"role": {"enum": ["admin", "user", "guest"]}},
} }
rc, stdout, stderr = self.run_schemagen(schema) rc, stdout, stderr = self.run_schemagen(schema)
@@ -82,7 +80,7 @@ class SchemagenKeywordTest(unittest.TestCase):
"module", "module",
"import", "import",
] ]
schema = {"type": "object", "additionalProperties": False, "properties": {}} schema = {"type": "object", "properties": {}}
for kw in keywords: for kw in keywords:
schema["properties"][kw] = {"type": "string"} schema["properties"][kw] = {"type": "string"}
rc, stdout, stderr = self.run_schemagen(schema) rc, stdout, stderr = self.run_schemagen(schema)
@@ -166,18 +164,11 @@ class SchemagenCollisionTest(unittest.TestCase):
def test_kind_enum_does_not_duplicate_arr0(self): def test_kind_enum_does_not_duplicate_arr0(self):
schema = { schema = {
"type": "object", "type": "object",
"additionalProperties": False,
"properties": { "properties": {
"arr": {"type": "array", "items": {"type": "string"}}, "arr": {"type": "array", "items": {"type": "string"}},
"obj": {"$ref": "#/$defs/Arr0"}, "obj": {"$ref": "#/$defs/Arr0"},
}, },
"$defs": { "$defs": {"Arr0": {"type": "object", "properties": {}}},
"Arr0": {
"type": "object",
"additionalProperties": False,
"properties": {},
}
},
} }
out = self.generate_and_compile(schema) out = self.generate_and_compile(schema)
self.assertIn("struct Arr0", out) self.assertIn("struct Arr0", out)
@@ -192,13 +183,7 @@ class SchemagenCollisionTest(unittest.TestCase):
schema = { schema = {
"type": "array", "type": "array",
"items": {"$ref": "#/$defs/Skip"}, "items": {"$ref": "#/$defs/Skip"},
"$defs": { "$defs": {"Skip": {"type": "object", "properties": {}}},
"Skip": {
"type": "object",
"additionalProperties": False,
"properties": {},
}
},
} }
out = self.generate_and_compile(schema) out = self.generate_and_compile(schema)
self.assertIn("struct Skip", out) self.assertIn("struct Skip", out)
@@ -214,7 +199,6 @@ class SchemagenCollisionTest(unittest.TestCase):
"""Regression test for issue #17: enum and array $defs must be reused.""" """Regression test for issue #17: enum and array $defs must be reused."""
schema = { schema = {
"type": "object", "type": "object",
"additionalProperties": False,
"properties": { "properties": {
"role1": {"$ref": "#/$defs/Role"}, "role1": {"$ref": "#/$defs/Role"},
"role2": {"$ref": "#/$defs/Role"}, "role2": {"$ref": "#/$defs/Role"},
@@ -268,14 +252,17 @@ class SchemagenAdditionalPropertiesTest(unittest.TestCase):
self.assertNotEqual(rc, 0) self.assertNotEqual(rc, 0)
self.assertIn("additionalProperties: true is not supported", stderr) self.assertIn("additionalProperties: true is not supported", stderr)
def test_additional_properties_absent_rejected(self): def test_additional_properties_absent_defaults_to_strict(self):
schema = { schema = {
"type": "object", "type": "object",
"properties": {"name": {"type": "string"}}, "properties": {"name": {"type": "string"}},
} }
rc, stdout, stderr = self.run_schemagen(schema) rc, stdout, stderr = self.run_schemagen(schema)
self.assertNotEqual(rc, 0) self.assertEqual(rc, 0, msg=stderr)
self.assertIn("additionalProperties is required for object schemas", stderr) # The generated parser should reject unknown keys. Verify the key-matching
# helper returns -1 for an unknown key and cbKeyData rejects it.
self.assertIn("int matchKey(Kind k, std::string_view key) const {", stdout)
self.assertNotIn("bool isStrict(Kind k) const", stdout)
def test_additional_properties_false_accepted(self): def test_additional_properties_false_accepted(self):
schema = { schema = {
@@ -323,7 +310,6 @@ class SchemagenCyclicArrayTest(unittest.TestCase):
def test_object_field_to_self_referential_array_rejected(self): def test_object_field_to_self_referential_array_rejected(self):
schema = { schema = {
"type": "object", "type": "object",
"additionalProperties": False,
"properties": {"items": {"$ref": "#/$defs/Items"}}, "properties": {"items": {"$ref": "#/$defs/Items"}},
"$defs": { "$defs": {
"Items": { "Items": {
@@ -340,7 +326,6 @@ class SchemagenCyclicArrayTest(unittest.TestCase):
def test_chain_of_array_refs_rejected(self): def test_chain_of_array_refs_rejected(self):
schema = { schema = {
"type": "object", "type": "object",
"additionalProperties": False,
"properties": {"x": {"$ref": "#/$defs/A"}}, "properties": {"x": {"$ref": "#/$defs/A"}},
"$defs": { "$defs": {
"A": {"type": "array", "items": {"$ref": "#/$defs/B"}}, "A": {"type": "array", "items": {"$ref": "#/$defs/B"}},
@@ -354,7 +339,6 @@ class SchemagenCyclicArrayTest(unittest.TestCase):
def test_non_recursive_array_refs_still_allowed(self): def test_non_recursive_array_refs_still_allowed(self):
schema = { schema = {
"type": "object", "type": "object",
"additionalProperties": False,
"properties": {"roles": {"$ref": "#/$defs/Roles"}}, "properties": {"roles": {"$ref": "#/$defs/Roles"}},
"$defs": { "$defs": {
"Role": {"enum": ["admin", "user"]}, "Role": {"enum": ["admin", "user"]},
@@ -455,12 +439,10 @@ class SchemagenCyclicNullableObjectTest(unittest.TestCase):
self.skipTest("C++ compiler not available") self.skipTest("C++ compiler not available")
schema = { schema = {
"type": "object", "type": "object",
"additionalProperties": False,
"properties": {"self": {"$ref": "#/$defs/Self"}}, "properties": {"self": {"$ref": "#/$defs/Self"}},
"$defs": { "$defs": {
"Self": { "Self": {
"type": ["object", "null"], "type": ["object", "null"],
"additionalProperties": False,
"properties": {"self": {"$ref": "#/$defs/Self"}}, "properties": {"self": {"$ref": "#/$defs/Self"}},
} }
}, },
@@ -610,13 +592,6 @@ class SchemagenIntegerBoundaryTest(unittest.TestCase):
("123.0", "WeaselJson_OK", 123), ("123.0", "WeaselJson_OK", 123),
("9e18", "WeaselJson_OK", 9000000000000000000), ("9e18", "WeaselJson_OK", 9000000000000000000),
("10e18", "WeaselJson_REJECT", 0), ("10e18", "WeaselJson_REJECT", 0),
# Issue #56: zero written with a negative exponent must be accepted.
("0e-1", "WeaselJson_OK", 0),
("0e-2", "WeaselJson_OK", 0),
("0.0e-2", "WeaselJson_OK", 0),
("-0e-2", "WeaselJson_OK", 0),
("0e-20", "WeaselJson_OK", 0),
("0.000e-5", "WeaselJson_OK", 0),
] ]
cases_src = self._build_cases_array("root", cases) cases_src = self._build_cases_array("root", cases)
harness = textwrap.dedent( harness = textwrap.dedent(
@@ -659,9 +634,6 @@ class SchemagenIntegerBoundaryTest(unittest.TestCase):
('{"age":2.0}', "WeaselJson_OK", 2), ('{"age":2.0}', "WeaselJson_OK", 2),
('{"age":0.0001e4}', "WeaselJson_OK", 1), ('{"age":0.0001e4}', "WeaselJson_OK", 1),
('{"age":0.001}', "WeaselJson_REJECT", 0), ('{"age":0.001}', "WeaselJson_REJECT", 0),
# Issue #56: zero written with a negative exponent must be accepted.
('{"age":0e-2}', "WeaselJson_OK", 0),
('{"age":-0e-20}', "WeaselJson_OK", 0),
( (
'{"age":-9223372036854775808.0}', '{"age":-9223372036854775808.0}',
"WeaselJson_OK", "WeaselJson_OK",
@@ -698,7 +670,6 @@ class SchemagenIntegerBoundaryTest(unittest.TestCase):
) )
schema = { schema = {
"type": "object", "type": "object",
"additionalProperties": False,
"properties": {"age": {"type": "integer"}}, "properties": {"age": {"type": "integer"}},
"required": ["age"], "required": ["age"],
} }
@@ -814,22 +785,14 @@ class SchemagenStringEscapeTest(unittest.TestCase):
def test_newline_in_property_key(self): def test_newline_in_property_key(self):
"""A JSON key containing a newline must become a valid C++ literal.""" """A JSON key containing a newline must become a valid C++ literal."""
schema = { schema = {"type": "object", "properties": {"a\nb": {"type": "string"}}}
"type": "object",
"additionalProperties": False,
"properties": {"a\nb": {"type": "string"}},
}
out = self.generate_and_compile(schema) out = self.generate_and_compile(schema)
self.assertIn('// "a\\nb"', out) self.assertIn('// "a\\nb"', out)
self.assertIn('if (key == "a\\nb") return 0;', out) self.assertIn('if (key == "a\\nb") return 0;', out)
def test_newline_in_enum_value(self): def test_newline_in_enum_value(self):
"""An enum value containing a newline must become a valid C++ literal.""" """An enum value containing a newline must become a valid C++ literal."""
schema = { schema = {"type": "object", "properties": {"x": {"enum": ["a\nb"]}}}
"type": "object",
"additionalProperties": False,
"properties": {"x": {"enum": ["a\nb"]}},
}
out = self.generate_and_compile(schema) out = self.generate_and_compile(schema)
self.assertIn('static constexpr const char *X_names[] = { "a\\nb" };', out) self.assertIn('static constexpr const char *X_names[] = { "a\\nb" };', out)
@@ -837,7 +800,6 @@ class SchemagenStringEscapeTest(unittest.TestCase):
"""Mixed control characters in an enum value must be escaped.""" """Mixed control characters in an enum value must be escaped."""
schema = { schema = {
"type": "object", "type": "object",
"additionalProperties": False,
"properties": { "properties": {
"x": {"enum": ["x\ny\rz\tw\vq\x00\x01"]}, "x": {"enum": ["x\ny\rz\tw\vq\x00\x01"]},
}, },
@@ -850,11 +812,7 @@ class SchemagenStringEscapeTest(unittest.TestCase):
def test_backslash_and_quote_still_escaped(self): def test_backslash_and_quote_still_escaped(self):
"""Existing escaping for backslash and double quote must remain correct.""" """Existing escaping for backslash and double quote must remain correct."""
schema = { schema = {"type": "object", "properties": {'a"b\\c': {"type": "string"}}}
"type": "object",
"additionalProperties": False,
"properties": {'a"b\\c': {"type": "string"}},
}
out = self.generate_and_compile(schema) out = self.generate_and_compile(schema)
self.assertIn('// "a\\"b\\\\c"', out) self.assertIn('// "a\\"b\\\\c"', out)
self.assertIn('if (key == "a\\"b\\\\c") return 0;', out) self.assertIn('if (key == "a\\"b\\\\c") return 0;', out)
+2 -8
View File
@@ -417,13 +417,7 @@ class Builder:
tobj = TObj(name) tobj = TObj(name)
if defname is not None: if defname is not None:
self._building[defname] = (tobj, nullable) self._building[defname] = (tobj, nullable)
ap = node.get("additionalProperties", None) ap = node.get("additionalProperties", False)
if ap is None:
raise GenError(
"additionalProperties is required for object schemas; set it "
"explicitly to false to reject unknown keys (absent "
"additionalProperties is not supported)"
)
if ap is True: if ap is True:
raise GenError("additionalProperties: true is not supported") raise GenError("additionalProperties: true is not supported")
if isinstance(ap, dict): if isinstance(ap, dict):
@@ -1064,10 +1058,10 @@ private:
++trim; ++trim;
}} }}
int64_t finalExp = exp - fracDigits + trim; int64_t finalExp = exp - fracDigits + trim;
if (finalExp < 0) return false;
size_t leadingZeros = 0; size_t leadingZeros = 0;
while (leadingZeros < digits.size() && digits[leadingZeros] == '0') ++leadingZeros; while (leadingZeros < digits.size() && digits[leadingZeros] == '0') ++leadingZeros;
if (leadingZeros == digits.size()) {{ out = 0; return true; }} if (leadingZeros == digits.size()) {{ out = 0; return true; }}
if (finalExp < 0) return false;
if (leadingZeros > 0) digits.erase(0, leadingZeros); if (leadingZeros > 0) digits.erase(0, leadingZeros);
constexpr uint64_t kMaxNeg = 9223372036854775808ULL; constexpr uint64_t kMaxNeg = 9223372036854775808ULL;
+1 -3
View File
@@ -38,8 +38,6 @@ enum WeaselJsonStatus {
WeaselJson_REJECT, WeaselJson_REJECT,
/** json is too deeply nested */ /** json is too deeply nested */
WeaselJson_OVERFLOW, WeaselJson_OVERFLOW,
/** Tried to call parse on a null parser */
WeaselJson_NULL,
}; };
typedef struct WeaselJsonParser WeaselJsonParser; typedef struct WeaselJsonParser WeaselJsonParser;
@@ -67,7 +65,7 @@ void WeaselJsonParser_destroy(WeaselJsonParser *parser);
/** Incrementally parse `len` more bytes starting at `buf`. `buf` may be /** Incrementally parse `len` more bytes starting at `buf`. `buf` may be
* modified. Call with `len` 0 to indicate end of data. `buf` may be null if * modified. Call with `len` 0 to indicate end of data. `buf` may be null if
* `len` is 0. `len` must not be negative; a negative length is treated as a * `len` is 0. `len` must not be negative; a negative length is treated as a
* rejected input. Returns WeaselJson_NULL if parser is null */ * rejected input. */
WeaselJsonStatus WeaselJsonParser_parse(WeaselJsonParser *parser, char *buf, WeaselJsonStatus WeaselJsonParser_parse(WeaselJsonParser *parser, char *buf,
int len); int len);
-3
View File
@@ -46,9 +46,6 @@ WeaselJsonParser_destroy(WeaselJsonParser *parser) {
__attribute__((visibility("default"))) WeaselJsonStatus __attribute__((visibility("default"))) WeaselJsonStatus
WeaselJsonParser_parse(WeaselJsonParser *parser, char *buf, int len) { WeaselJsonParser_parse(WeaselJsonParser *parser, char *buf, int len) {
if (parser == nullptr) [[unlikely]] {
return WeaselJson_NULL;
}
return ((Parser3 *)parser)->parse(buf, len); return ((Parser3 *)parser)->parse(buf, len);
} }
} }
+20 -10
View File
@@ -139,7 +139,8 @@ struct Parser3 {
stackPtr = stack(); stackPtr = stack();
std::ignore = push({N_VALUE, N_WHITESPACE, T_EOF}); std::ignore = push({N_VALUE, N_WHITESPACE, T_EOF});
inKey = false; inKey = false;
terminalStatus = WeaselJson_OK; rejected = false;
overflowed = false;
utf8Codepoint = 0; utf8Codepoint = 0;
utf16Surrogate = 0; utf16Surrogate = 0;
minCodepoint = 0; minCodepoint = 0;
@@ -162,7 +163,8 @@ struct Parser3 {
NumDfa numDfa; NumDfa numDfa;
Utf8Dfa strDfa; Utf8Dfa strDfa;
bool inKey = false; bool inKey = false;
WeaselJsonStatus terminalStatus = WeaselJson_OK; bool rejected = false;
bool overflowed = false;
#ifndef HAS_MUSTTAIL #ifndef HAS_MUSTTAIL
char *stashBufForTrampoline; char *stashBufForTrampoline;
@@ -650,7 +652,7 @@ inline PRESERVE_NONE ContinuationStatus n_string2(Parser3 *self, char *buf,
self->writeBuf[0] = (0b00000111 & codepoint) | 0b11110000; self->writeBuf[0] = (0b00000111 & codepoint) | 0b11110000;
self->writeBuf += 4; self->writeBuf += 4;
} }
} else if (0xdc00 <= codepoint && codepoint <= 0xdfff) [[unlikely]] { } else if (0xdc00 <= codepoint && codepoint <= 0xdfff) {
return WeaselJson_REJECT; return WeaselJson_REJECT;
} else { } else {
if (!(self->flags & WeaselJsonRaw)) { if (!(self->flags & WeaselJsonRaw)) {
@@ -1070,12 +1072,16 @@ constexpr inline struct ContinuationTable {
inline WeaselJsonStatus Parser3::parse(char *buf, int len) { inline WeaselJsonStatus Parser3::parse(char *buf, int len) {
this->dataBegin = this->writeBuf = buf; this->dataBegin = this->writeBuf = buf;
if (this->terminalStatus != WeaselJson_OK) [[unlikely]] { if (this->rejected) [[unlikely]] {
return this->terminalStatus; return WeaselJson_REJECT;
}
if (this->overflowed) [[unlikely]] {
return WeaselJson_OVERFLOW;
} }
if (len < 0) [[unlikely]] { if (len < 0) [[unlikely]] {
this->terminalStatus = WeaselJson_REJECT; this->rejected = true;
return WeaselJson_REJECT; return WeaselJson_REJECT;
} }
@@ -1085,8 +1091,10 @@ inline WeaselJsonStatus Parser3::parse(char *buf, int len) {
// range. // range.
ContinuationStatus status = ContinuationStatus status =
symbolTables.continuations[top()](this, buf, buf + len); symbolTables.continuations[top()](this, buf, buf + len);
if (status > WeaselJson_AGAIN) { if (status == WeaselJson_REJECT) {
this->terminalStatus = WeaselJsonStatus(status); this->rejected = true;
} else if (status == WeaselJson_OVERFLOW) {
this->overflowed = true;
} }
return WeaselJsonStatus(status); return WeaselJsonStatus(status);
#else #else
@@ -1095,8 +1103,10 @@ inline WeaselJsonStatus Parser3::parse(char *buf, int len) {
while ((result = symbolTables.continuations[top()]( while ((result = symbolTables.continuations[top()](
this, stashBufForTrampoline, buf + len)) == kBounce) this, stashBufForTrampoline, buf + len)) == kBounce)
; ;
if (result > WeaselJson_AGAIN) { if (result == WeaselJson_REJECT) {
this->terminalStatus = WeaselJsonStatus(result); this->rejected = true;
} else if (result == WeaselJson_OVERFLOW) {
this->overflowed = true;
} }
return WeaselJsonStatus(result); return WeaselJsonStatus(result);
#endif #endif
-4
View File
@@ -330,10 +330,6 @@ TEST_CASE("parse rejects negative length") {
WeaselJsonParser_destroy(parser); WeaselJsonParser_destroy(parser);
} }
TEST_CASE("Calling parse with nullptr doesn't crash") {
REQUIRE(WeaselJsonParser_parse(nullptr, nullptr, 0) == WeaselJson_NULL);
}
TEST_CASE("streaming") { testStreaming(json); } TEST_CASE("streaming") { testStreaming(json); }
TEST_CASE("reset clears inKey and transient state") { TEST_CASE("reset clears inKey and transient state") {
-3
View File
@@ -33,9 +33,6 @@ int main(int argc, char **argv) {
case WeaselJson_REJECT: case WeaselJson_REJECT:
case WeaselJson_OVERFLOW: case WeaselJson_OVERFLOW:
return 1; return 1;
case WeaselJson_NULL:
fprintf(stderr, "parse called with a null parser\n");
return 1;
} }
if (l == 0) { if (l == 0) {
return 1; return 1;
-1
View File
@@ -30,7 +30,6 @@ class WeaselJsonStatus(enum.Enum):
AGAIN = 1 AGAIN = 1
REJECT = 2 REJECT = 2
OVERFLOW = 3 OVERFLOW = 3
NULL = 4
class WeaselJsonCallbacksBase: class WeaselJsonCallbacksBase: