Files
reasampler/src/core/json/json.cpp
T

313 lines
9.4 KiB
C++
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// core/json implementation — see json.h. Any behavioral change here changes
// every persisted-blob parser that shares this lexical layer at once.
#include "core/json/json.h"
#include <cerrno>
#include <climits>
#include <cstdio>
#include <cstdlib>
namespace reasampler::json {
// -- emit helpers -------------------------------------------------------
void writeEscaped(std::string& out, const std::string& s) {
out += '"';
for (char c : s) {
switch (c) {
case '"': out += "\\\""; break;
case '\\': out += "\\\\"; break;
case '\b': out += "\\b"; break;
case '\f': out += "\\f"; break;
case '\n': out += "\\n"; break;
case '\r': out += "\\r"; break;
case '\t': out += "\\t"; break;
default:
if (static_cast<unsigned char>(c) < 0x20) {
char buf[8];
std::snprintf(buf, sizeof(buf), "\\u%04x", static_cast<unsigned char>(c));
out += buf;
} else {
out += c;
}
}
}
out += '"';
}
std::string numToStr(double v) {
char buf[32];
std::snprintf(buf, sizeof(buf), "%.17g", v);
return buf;
}
std::string numToStr(std::int64_t v) {
char buf[32];
std::snprintf(buf, sizeof(buf), "%lld", static_cast<long long>(v));
return buf;
}
std::string numToStr(int v) {
char buf[16];
std::snprintf(buf, sizeof(buf), "%d", v);
return buf;
}
void writeStringArray(std::string& out, const std::vector<std::string>& v) {
out += '[';
for (std::size_t i = 0; i < v.size(); ++i) {
if (i) out += ',';
writeEscaped(out, v[i]);
}
out += ']';
}
void writeIntArray(std::string& out, const std::vector<int>& v) {
out += '[';
for (std::size_t i = 0; i < v.size(); ++i) {
if (i) out += ',';
out += numToStr(v[i]);
}
out += ']';
}
// -- Reader ---------------------------------------------------------------
void Reader::skipWs() {
while (!eof()) {
char c = s_[pos_];
if (c == ' ' || c == '\t' || c == '\n' || c == '\r') ++pos_;
else break;
}
}
bool Reader::consume(char c) {
skipWs();
if (eof() || s_[pos_] != c) return false;
++pos_;
return true;
}
// Parses a JSON string literal (with the escapes our writers emit, plus \uXXXX
// for control chars). Positioned before the opening quote (skips leading ws).
bool Reader::parseString(std::string& out) {
skipWs();
if (eof() || s_[pos_] != '"') return false;
++pos_;
out.clear();
while (!eof()) {
char c = s_[pos_++];
if (c == '"') return true;
if (c == '\\') {
if (eof()) return false;
char e = s_[pos_++];
switch (e) {
case '"': out += '"'; break;
case '\\': out += '\\'; break;
case '/': out += '/'; break;
case 'b': out += '\b'; break;
case 'f': out += '\f'; break;
case 'n': out += '\n'; break;
case 'r': out += '\r'; break;
case 't': out += '\t'; break;
case 'u': {
auto readHex4 = [&](unsigned int& cp) -> bool {
if (pos_ + 4 > s_.size()) return false;
cp = 0;
for (int i = 0; i < 4; ++i) {
char h = s_[pos_++];
cp <<= 4;
if (h >= '0' && h <= '9') cp |= static_cast<unsigned>(h - '0');
else if (h >= 'a' && h <= 'f') cp |= static_cast<unsigned>(h - 'a' + 10);
else if (h >= 'A' && h <= 'F') cp |= static_cast<unsigned>(h - 'A' + 10);
else return false;
}
return true;
};
unsigned int hi = 0;
if (!readHex4(hi)) return false;
unsigned int codePoint = hi;
if (hi >= 0xD800 && hi <= 0xDBFF) {
// High surrogate — must be followed by \uDC00\uDFFF.
if (pos_ + 6 > s_.size()) return false;
if (s_[pos_] != '\\' || s_[pos_ + 1] != 'u') return false;
pos_ += 2;
unsigned int lo = 0;
if (!readHex4(lo)) return false;
if (lo < 0xDC00 || lo > 0xDFFF) return false; // unpaired high surrogate
codePoint = 0x10000 + ((hi - 0xD800) << 10) + (lo - 0xDC00);
} else if (hi >= 0xDC00 && hi <= 0xDFFF) {
return false; // unpaired low surrogate — malformed
}
if (codePoint <= 0x7F) {
out += static_cast<char>(codePoint);
} else if (codePoint <= 0x7FF) {
out += static_cast<char>(0xC0 | (codePoint >> 6));
out += static_cast<char>(0x80 | (codePoint & 0x3F));
} else if (codePoint <= 0xFFFF) {
out += static_cast<char>(0xE0 | (codePoint >> 12));
out += static_cast<char>(0x80 | ((codePoint >> 6) & 0x3F));
out += static_cast<char>(0x80 | (codePoint & 0x3F));
} else {
out += static_cast<char>(0xF0 | (codePoint >> 18));
out += static_cast<char>(0x80 | ((codePoint >> 12) & 0x3F));
out += static_cast<char>(0x80 | ((codePoint >> 6) & 0x3F));
out += static_cast<char>(0x80 | (codePoint & 0x3F));
}
break;
}
default: return false;
}
} else {
out += c;
}
}
return false; // unterminated string
}
bool Reader::parseRawScalar(std::string& out) {
skipWs();
std::size_t start = pos_;
while (!eof()) {
char c = s_[pos_];
if (c == ',' || c == '}' || c == ']' || c == ' ' || c == '\t' ||
c == '\n' || c == '\r')
break;
++pos_;
}
if (pos_ == start) return false;
out.assign(s_, start, pos_ - start);
return true;
}
bool Reader::parseDouble(double& out) {
std::string tok;
if (!parseRawScalar(tok)) return false;
const char* b = tok.c_str();
char* end = nullptr;
errno = 0;
double v = std::strtod(b, &end);
if (end != b + tok.size()) return false;
if (errno == ERANGE) return false; // overflow / underflow -> malformed
out = v;
return true;
}
bool Reader::parseInt64(std::int64_t& out) {
std::string tok;
if (!parseRawScalar(tok)) return false;
const char* b = tok.c_str();
char* end = nullptr;
errno = 0;
long long v = std::strtoll(b, &end, 10);
if (end != b + tok.size()) return false;
if (errno == ERANGE) return false; // overflow -> malformed
out = static_cast<std::int64_t>(v);
return true;
}
bool Reader::parseInt(int& out) {
std::int64_t v = 0;
if (!parseInt64(v)) return false;
if (v < INT_MIN || v > INT_MAX) return false;
out = static_cast<int>(v);
return true;
}
bool Reader::parseBool(bool& out) {
std::string tok;
if (!parseRawScalar(tok)) return false;
if (tok == "true") { out = true; return true; }
if (tok == "false") { out = false; return true; }
return false;
}
bool Reader::expectNullOr(bool& wasNull) {
skipWs();
if (eof()) return false;
if (s_.compare(pos_, 4, "null") == 0) {
pos_ += 4;
wasNull = true;
} else {
wasNull = false;
}
return true;
}
bool Reader::parseKey(std::string& key) {
if (!parseString(key)) return false;
return consume(':');
}
bool Reader::parseStringArray(std::vector<std::string>& out) {
if (!consume('[')) return false;
skipWs();
if (consume(']')) return true; // empty array
do {
std::string s;
if (!parseString(s)) return false;
out.push_back(std::move(s));
} while (consume(','));
return consume(']');
}
bool Reader::parseIntArray(std::vector<int>& out) {
if (!consume('[')) return false;
skipWs();
if (consume(']')) return true;
do {
int v = 0;
if (!parseInt(v)) return false;
out.push_back(v);
} while (consume(','));
return consume(']');
}
bool Reader::skipValue() {
std::string raw;
return captureValue(raw);
}
// Records the raw source span of one JSON value starting at the current position
// (after whitespace). Handles nested objects/arrays with string-aware brace
// matching (braces inside strings ignored).
bool Reader::captureValue(std::string& raw) {
skipWs();
if (eof()) return false;
std::size_t start = pos_;
char c = s_[pos_];
if (c == '"') {
std::string tmp;
if (!parseString(tmp)) return false;
raw.assign(s_, start, pos_ - start);
return true;
}
if (c == '{' || c == '[') {
char open = c, close = (c == '{') ? '}' : ']';
++pos_;
int depth = 1;
while (!eof() && depth > 0) {
char d = s_[pos_];
if (d == '"') {
std::string tmp;
if (!parseString(tmp)) return false; // advances past the string
continue;
}
if (d == open) ++depth;
else if (d == close) --depth;
++pos_;
}
if (depth != 0) return false;
raw.assign(s_, start, pos_ - start);
return true;
}
// bare scalar (number / true / false / null)
return parseRawScalar(raw);
}
} // namespace reasampler::json