schemagen: deduplicate enum constants that collide after sanitization

Distinct JSON enum values can sanitize to the same C++ identifier
(e.g. "foo-bar" and "foo_bar" both become `foo_bar`), producing an
invalid `enum class` with duplicate constants.

Add `unique_enum_identifiers()` which appends a numeric suffix to
later collisions while preserving enum declaration order, so the
index-to-JSON-value mapping used by the generated parser stays intact.

Also add contrib/schemagen/test_schemagen.py and wire the schemagen
Python tests plus the existing example.schema.json/test_gen.cpp example
into ctest via CMakeLists.txt.

Closes #4
This commit is contained in:
2026-06-18 11:02:47 -04:00
parent 7a8e5f84f2
commit 359f4f4bb6
3 changed files with 114 additions and 1 deletions
+32
View File
@@ -170,6 +170,38 @@ if(Python3_Interpreter_FOUND)
${CMAKE_CURRENT_SOURCE_DIR}/test_python_bindings.py)
endif()
# schemagen tests
add_test(
NAME schemagen_python_tests
COMMAND python3 contrib/schemagen/test_schemagen.py
WORKING_DIRECTORY ${CMAKE_SOURCE_DIR})
set(SCHEMAGEN_DIR ${CMAKE_SOURCE_DIR}/contrib/schemagen)
set(SCHEMAGEN_SCRIPT ${SCHEMAGEN_DIR}/weaseljson_schemagen.py)
set(EXAMPLE_SCHEMA ${SCHEMAGEN_DIR}/example.schema.json)
set(GEN_H ${CMAKE_BINARY_DIR}/gen.h)
add_custom_command(
OUTPUT ${GEN_H}
COMMAND python3 ${SCHEMAGEN_SCRIPT} ${EXAMPLE_SCHEMA} -o ${GEN_H} --namespace
weasel_schema
DEPENDS ${SCHEMAGEN_SCRIPT} ${EXAMPLE_SCHEMA}
COMMENT "Generating gen.h from example.schema.json")
add_custom_target(schemagen_gen_h DEPENDS ${GEN_H})
add_executable(schemagen_example ${SCHEMAGEN_DIR}/test_gen.cpp)
target_include_directories(schemagen_example PRIVATE include
${CMAKE_BINARY_DIR})
target_link_libraries(schemagen_example PRIVATE ${PROJECT_NAME})
target_compile_options(schemagen_example PRIVATE -Wno-switch-enum)
add_dependencies(schemagen_example schemagen_gen_h)
add_test(
NAME schemagen_example
COMMAND schemagen_example
WORKING_DIRECTORY ${CMAKE_BINARY_DIR})
include(CMakePushCheckState)
include(CheckCXXCompilerFlag)
cmake_push_check_state()
+53
View File
@@ -0,0 +1,53 @@
#!/usr/bin/env python3
"""Tests for weaseljson_schemagen.py."""
import json
import os
import subprocess
import sys
import tempfile
import unittest
SCRIPT = os.path.join(os.path.dirname(__file__), "weaseljson_schemagen.py")
class SchemagenEnumTest(unittest.TestCase):
def run_schemagen(self, schema, args=None):
"""Run schemagen on a schema dict. Returns (returncode, stdout, stderr)."""
with tempfile.NamedTemporaryFile("w", suffix=".json", delete=False) as fp:
json.dump(schema, fp)
schema_path = fp.name
try:
cmd = [sys.executable, SCRIPT, schema_path]
if args:
cmd.extend(args)
result = subprocess.run(cmd, capture_output=True, text=True, check=False)
return result.returncode, result.stdout, result.stderr
finally:
os.unlink(schema_path)
def test_colliding_enum_values_deduplicate(self):
schema = {
"type": "object",
"properties": {"role": {"enum": ["foo-bar", "foo_bar"]}},
}
rc, stdout, stderr = self.run_schemagen(schema)
self.assertEqual(rc, 0, msg=stderr)
self.assertIn("enum class Role : int { foo_bar, foo_bar_1 };", stdout)
self.assertIn(
'static constexpr const char *Role_names[] = { "foo-bar", "foo_bar" };',
stdout,
)
def test_distinct_enum_values_generate(self):
schema = {
"type": "object",
"properties": {"role": {"enum": ["admin", "user", "guest"]}},
}
rc, stdout, stderr = self.run_schemagen(schema)
self.assertEqual(rc, 0, msg=stderr)
self.assertIn("enum class Role : int { admin, user, guest };", stdout)
if __name__ == "__main__":
unittest.main()
+29 -1
View File
@@ -86,6 +86,34 @@ def sanitize(name, fallback="x"):
return s
def unique_enum_identifiers(values):
"""Return a list of sanitized C++ identifiers, one per input value.
Distinct JSON enum values may sanitize to the same C++ token (e.g.
"foo-bar" and "foo_bar" both become "foo_bar"). This helper appends a
numeric suffix to later collisions so the generated `enum class` stays
valid while preserving the original order and therefore the index-to-value
mapping used at parse time.
"""
used = set()
out = []
for v in values:
base = sanitize(v)
if base not in used:
used.add(base)
out.append(base)
continue
n = 1
while True:
candidate = f"{base}_{n}"
if candidate not in used:
used.add(candidate)
out.append(candidate)
break
n += 1
return out
def camel(name):
parts = [p for p in name.replace("-", " ").replace("_", " ").split(" ") if p]
if not parts:
@@ -508,7 +536,7 @@ namespace {ns} {{"""
return ""
out = []
for e in self.b.enums.values():
vals = ", ".join(sanitize(v) for v in e.values)
vals = ", ".join(unique_enum_identifiers(e.values))
out.append(f"enum class {e.name} : int {{ {vals} }};")
return "\n".join(out) + "\n"