diff --git a/src/to_json.cpp b/src/to_json.cpp index eb76ee6a..73cad8d3 100644 --- a/src/to_json.cpp +++ b/src/to_json.cpp @@ -128,12 +128,25 @@ class JsonWriter { else if (opt.quote_numbers < 2 && is_numb(value) && // exception: 012 (but not 0.12) is assumed to be a string (value[0] != '0' || value[1] == '.' || value[1] == '\0') && + // exception: digit-string + E + digit-string (e.g. 6E72) is + // assumed to be an identifier such as a PDB code, not a number + !looks_like_numberlike_id(value) && (opt.quote_numbers == 0 || value.back() != ')')) write_as_number(value); else write_string(as_string(value)); } + // returns true for tokens of the form digits + E + digits (e.g. 6E72): + // valid CIF numbers but indistinguishable from alphanumeric identifiers + // such as PDB codes, which a JSON reader would parse as huge floats + static bool looks_like_numberlike_id(const std::string& v) { + size_t epos = v.find('E'); + return epos != std::string::npos && + v.find_first_not_of("0123456789") == epos && + v.find_first_not_of("0123456789", epos + 1) == std::string::npos; + } + void open_cat(const std::string& cat, size_t* tag_pos) { if (!cat.empty()) { change_indent(+1); diff --git a/tests/test_cif_json.py b/tests/test_cif_json.py index f3931008..5c92f5b2 100644 --- a/tests/test_cif_json.py +++ b/tests/test_cif_json.py @@ -52,6 +52,13 @@ def test_json_number_normalization(self): self.assertEqual(parsed, {'test': {'_a': -0.99, '_b': 0.04, '_c': -0.04, '_d': 0.5}}) + def test_json_numberlike_id(self): + # 6E72 is a valid CIF number but also a PDB code - quote it (#444) + doc = cif.read_string('data_6E72 _entry.id 6E72 _a 1.2E3 _b 2e4') + parsed = json.loads(doc.as_json()) + self.assertEqual(parsed, {'6e72': {'_entry.id': '6E72', + '_a': 1200.0, '_b': 20000.0}}) + class TestMmjson(unittest.TestCase): def test_read_1pfe(self): path = full_path('1pfe.json')