Public Access
A streaming JSON scanner (a 33 KB list of releases costs a few hundred bytes), the release reader built on it, HTTP response heads and chunked bodies, URLs, and the decisions: which release is an update, whether to announce it, and which download URLs the device takes. Tested against the real answers of git.twis.la. Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01EhqxQ49eCju4CzKYNjZzwT
203 lines
6.0 KiB
C++
203 lines
6.0 KiB
C++
#include "json_scan.h"
|
|
|
|
namespace roro::release {
|
|
|
|
namespace {
|
|
bool space(char c) { return c == ' ' || c == '\t' || c == '\r' || c == '\n'; }
|
|
bool continuation(char c) { return (static_cast<uint8_t>(c) & 0xC0) == 0x80; }
|
|
} // namespace
|
|
|
|
void JsonScanner::feed(const char* data, size_t len) {
|
|
for (size_t i = 0; i < len && !failed_; i++) step(data[i]);
|
|
}
|
|
|
|
std::string JsonScanner::path() const {
|
|
std::string p;
|
|
for (const Frame& f : stack_) {
|
|
if (f.isObject) {
|
|
if (!p.empty()) p += '.';
|
|
p += f.key;
|
|
} else {
|
|
p += '[' + std::to_string(f.index) + ']';
|
|
}
|
|
}
|
|
return p;
|
|
}
|
|
|
|
void JsonScanner::append(const char* bytes, size_t n) {
|
|
if (text_.size() + n > max_) {
|
|
truncated_ = true;
|
|
return;
|
|
}
|
|
text_.append(bytes, n);
|
|
}
|
|
|
|
void JsonScanner::appendCodePoint(uint32_t cp) {
|
|
char b[4];
|
|
size_t n;
|
|
if (cp < 0x80) {
|
|
b[0] = static_cast<char>(cp);
|
|
n = 1;
|
|
} else if (cp < 0x800) {
|
|
b[0] = static_cast<char>(0xC0 | (cp >> 6));
|
|
b[1] = static_cast<char>(0x80 | (cp & 0x3F));
|
|
n = 2;
|
|
} else if (cp < 0x10000) {
|
|
b[0] = static_cast<char>(0xE0 | (cp >> 12));
|
|
b[1] = static_cast<char>(0x80 | ((cp >> 6) & 0x3F));
|
|
b[2] = static_cast<char>(0x80 | (cp & 0x3F));
|
|
n = 3;
|
|
} else {
|
|
b[0] = static_cast<char>(0xF0 | (cp >> 18));
|
|
b[1] = static_cast<char>(0x80 | ((cp >> 12) & 0x3F));
|
|
b[2] = static_cast<char>(0x80 | ((cp >> 6) & 0x3F));
|
|
b[3] = static_cast<char>(0x80 | (cp & 0x3F));
|
|
n = 4;
|
|
}
|
|
append(b, n);
|
|
}
|
|
|
|
void JsonScanner::startString(bool key) {
|
|
inString_ = true;
|
|
isKey_ = key;
|
|
escape_ = false;
|
|
unicodeLeft_ = 0;
|
|
highSurrogate_ = 0;
|
|
truncated_ = false;
|
|
text_.clear();
|
|
}
|
|
|
|
void JsonScanner::stringChar(char c) {
|
|
if (unicodeLeft_ > 0) {
|
|
int digit = c >= '0' && c <= '9' ? c - '0' : c >= 'a' && c <= 'f' ? c - 'a' + 10 : c >= 'A' && c <= 'F' ? c - 'A' + 10 : -1;
|
|
if (digit < 0) return fail();
|
|
unicode_ = unicode_ * 16 + digit;
|
|
if (--unicodeLeft_ == 0) {
|
|
if (unicode_ >= 0xD800 && unicode_ < 0xDC00) {
|
|
highSurrogate_ = unicode_; // the low half comes next
|
|
} else if (unicode_ >= 0xDC00 && unicode_ < 0xE000 && highSurrogate_) {
|
|
appendCodePoint(0x10000 + ((highSurrogate_ - 0xD800) << 10) + (unicode_ - 0xDC00));
|
|
highSurrogate_ = 0;
|
|
} else {
|
|
appendCodePoint(unicode_);
|
|
highSurrogate_ = 0;
|
|
}
|
|
}
|
|
return;
|
|
}
|
|
if (escape_) {
|
|
escape_ = false;
|
|
switch (c) {
|
|
case 'n': append("\n", 1); break;
|
|
case 't': append("\t", 1); break;
|
|
case 'r': append("\r", 1); break;
|
|
case 'b': append("\b", 1); break;
|
|
case 'f': append("\f", 1); break;
|
|
case 'u': unicodeLeft_ = 4; unicode_ = 0; break;
|
|
default: append(&c, 1); break; // \" \\ \/
|
|
}
|
|
return;
|
|
}
|
|
if (c == '\\') {
|
|
escape_ = true;
|
|
} else if (c == '"') {
|
|
endString();
|
|
} else {
|
|
append(&c, 1);
|
|
}
|
|
}
|
|
|
|
void JsonScanner::endString() {
|
|
inString_ = false;
|
|
// A cut can leave half a character at the end.
|
|
while (truncated_ && !text_.empty()) {
|
|
size_t k = text_.size();
|
|
while (k > 0 && continuation(text_[k - 1])) k--;
|
|
if (k == 0) break;
|
|
uint8_t lead = static_cast<uint8_t>(text_[k - 1]);
|
|
size_t want = lead >= 0xF0 ? 4 : lead >= 0xE0 ? 3 : lead >= 0xC0 ? 2 : 1;
|
|
if (text_.size() - (k - 1) < want) text_.resize(k - 1);
|
|
break;
|
|
}
|
|
if (isKey_) {
|
|
stack_.back().key = text_;
|
|
expect_ = Expect::Colon;
|
|
return;
|
|
}
|
|
sink_(path(), text_, true, truncated_);
|
|
valueDone();
|
|
}
|
|
|
|
void JsonScanner::endLiteral() {
|
|
inLiteral_ = false;
|
|
sink_(path(), text_, false, false);
|
|
valueDone();
|
|
}
|
|
|
|
void JsonScanner::valueDone() {
|
|
if (stack_.empty()) {
|
|
done_ = true;
|
|
return;
|
|
}
|
|
expect_ = Expect::CommaOrEnd;
|
|
}
|
|
|
|
void JsonScanner::step(char c) {
|
|
if (inString_) return stringChar(c);
|
|
if (inLiteral_) {
|
|
if (space(c) || c == ',' || c == '}' || c == ']') {
|
|
endLiteral(); // and the character that ended it is read again below
|
|
} else {
|
|
if (text_.size() < max_) text_ += c;
|
|
return;
|
|
}
|
|
}
|
|
if (space(c)) return;
|
|
switch (expect_) {
|
|
case Expect::Value:
|
|
if (c == '{') {
|
|
stack_.push_back({true, "", 0});
|
|
expect_ = Expect::KeyOrEnd;
|
|
} else if (c == '[') {
|
|
stack_.push_back({false, "", 0});
|
|
expect_ = Expect::Value;
|
|
} else if (c == ']' && !stack_.empty() && !stack_.back().isObject && stack_.back().index == 0) {
|
|
stack_.pop_back(); // an empty array
|
|
valueDone();
|
|
} else if (c == '"') {
|
|
startString(false);
|
|
} else if (c == '-' || (c >= '0' && c <= '9') || c == 't' || c == 'f' || c == 'n') {
|
|
inLiteral_ = true;
|
|
text_.assign(1, c);
|
|
} else {
|
|
fail();
|
|
}
|
|
return;
|
|
case Expect::KeyOrEnd:
|
|
if (c == '"') startString(true);
|
|
else if (c == '}') {
|
|
stack_.pop_back();
|
|
valueDone();
|
|
} else fail();
|
|
return;
|
|
case Expect::Colon:
|
|
if (c == ':') expect_ = Expect::Value;
|
|
else fail();
|
|
return;
|
|
case Expect::CommaOrEnd:
|
|
if (c == ',') {
|
|
if (stack_.back().isObject) expect_ = Expect::KeyOrEnd;
|
|
else {
|
|
stack_.back().index++;
|
|
expect_ = Expect::Value;
|
|
}
|
|
} else if ((c == '}' && stack_.back().isObject) || (c == ']' && !stack_.back().isObject)) {
|
|
stack_.pop_back();
|
|
valueDone();
|
|
} else fail();
|
|
return;
|
|
}
|
|
}
|
|
|
|
} // namespace roro::release
|