mirror of
https://github.com/Cockatrice/Cockatrice.git
synced 2026-09-21 09:05:10 -07:00
[Oracle] Parse sets lazily to slash importer peak memory (#7217)
* [Oracle] Parse sets lazily to slash importer peak memory - Add a raw JSON scanner that splits the document into per-set byte ranges without materializing the JSON tree - Keep only the raw document bytes and parse one set at a time in startImport - Take readSetsFromByteArray by value so the wizard's buffer is moved, not copied - Clear the retained raw data in releaseSetData()/clear() - Cover the scanner and lazy parsing with tests Took 2 minutes * [Oracle] Fix nesting-depth cap, tolerate unescaped control chars, lazy-parse review fixes --------- Co-authored-by: Lukas Brübach <Bruebach.Lukas@bdosecurity.de>
This commit is contained in:
parent
1dc54617ba
commit
c011ea7ceb
8 changed files with 992 additions and 45 deletions
|
|
@ -7,6 +7,7 @@
|
|||
#include <QDebug>
|
||||
#include <QJsonDocument>
|
||||
#include <QJsonObject>
|
||||
#include <QJsonParseError>
|
||||
#include <QRegularExpression>
|
||||
#include <QSet>
|
||||
#include <algorithm>
|
||||
|
|
@ -44,26 +45,23 @@ static CardSet::Priority getSetPriority(const QString &setType, const QString &s
|
|||
return priority;
|
||||
}
|
||||
|
||||
bool OracleImporter::readSetsFromByteArray(const QByteArray &data)
|
||||
bool OracleImporter::readSetsFromByteArray(QByteArray data)
|
||||
{
|
||||
QJsonParseError error;
|
||||
auto doc = QJsonDocument::fromJson(data, &error);
|
||||
if (error.error != QJsonParseError::NoError) {
|
||||
qDebug() << "error: QJsonDocument::fromJson():" << error.errorString();
|
||||
RawJson::ScanError error;
|
||||
const QList<RawJson::SetRange> ranges = RawJson::scanSetRanges(data, &error);
|
||||
if (error.isError()) {
|
||||
qDebug() << "error: RawJson::scanSetRanges():" << error.message;
|
||||
return false;
|
||||
}
|
||||
|
||||
auto setsObj = doc.object().value("data").toObject();
|
||||
|
||||
QList<SetToDownload> newSetList;
|
||||
newSetList.reserve(ranges.size());
|
||||
|
||||
for (auto it = setsObj.constBegin(); it != setsObj.constEnd(); ++it) {
|
||||
QJsonObject setObj = it.value().toObject();
|
||||
QString shortName = setObj.value("code").toString().toUpper();
|
||||
QString longName = setObj.value("name").toString();
|
||||
QJsonArray setCards = setObj.value("cards").toArray();
|
||||
QString setType = setObj.value("type").toString();
|
||||
QDate releaseDate = QDate::fromString(setObj.value("releaseDate").toString(), Qt::ISODate);
|
||||
for (const RawJson::SetRange &range : ranges) {
|
||||
QString shortName = range.code.toUpper();
|
||||
QString longName = range.name;
|
||||
QString setType = range.type;
|
||||
QDate releaseDate = QDate::fromString(range.releaseDate, Qt::ISODate);
|
||||
CardSet::Priority priority = getSetPriority(setType, shortName);
|
||||
// capitalize set type
|
||||
if (setType.length() > 0) {
|
||||
|
|
@ -83,7 +81,9 @@ bool OracleImporter::readSetsFromByteArray(const QByteArray &data)
|
|||
}
|
||||
setType = setType.trimmed();
|
||||
}
|
||||
newSetList.append(SetToDownload(shortName, longName, setCards, priority, setType, releaseDate));
|
||||
SetToDownload set(shortName, longName, priority, setType, releaseDate);
|
||||
set.setRawRange(range.dataRange);
|
||||
newSetList.append(set);
|
||||
}
|
||||
|
||||
std::sort(newSetList.begin(), newSetList.end());
|
||||
|
|
@ -92,6 +92,7 @@ bool OracleImporter::readSetsFromByteArray(const QByteArray &data)
|
|||
return false;
|
||||
}
|
||||
allSets = newSetList;
|
||||
rawSetsData = std::move(data);
|
||||
return true;
|
||||
}
|
||||
|
||||
|
|
@ -550,22 +551,16 @@ int OracleImporter::startImport()
|
|||
{
|
||||
static ICardSetPriorityController *noOpController = new NoopCardSetPriorityController();
|
||||
|
||||
// Pre-allocate the cards hash to avoid rehashing during import. The hash
|
||||
// is keyed by distinct card name rather than by printings: AllPrintings
|
||||
// ships ~100k printings for ~35k names, so reserving the printing count
|
||||
// would overallocate ~3x (against this stack's RAM goal). Collecting
|
||||
// distinct names is cheap — one pass over the already-parsed name fields.
|
||||
{
|
||||
QSet<QString> distinctNames;
|
||||
for (const SetToDownload &curSetToParse : allSets) {
|
||||
for (const QJsonValue &cardValue : curSetToParse.getCards()) {
|
||||
distinctNames.insert(cardValue.toObject().value("name").toString());
|
||||
}
|
||||
}
|
||||
cards.reserve(distinctNames.size());
|
||||
// The set goes out of scope here, handing the ~35k name QStrings back
|
||||
// to the allocator before the (memory-heavy) import loop starts.
|
||||
// Pre-allocate the cards hash to avoid rehashing during import. Keys are
|
||||
// distinct card names while raw ranges only count printings (AllPrintings
|
||||
// ~100k printings vs ~35k names), so this over-reserves somewhat; an exact
|
||||
// distinct-name count would require eagerly parsing, which the lazy reader
|
||||
// deliberately avoids. It's a capacity hint, so the overshoot is harmless.
|
||||
int estimatedCards = 0;
|
||||
for (const SetToDownload &curSetToParse : allSets) {
|
||||
estimatedCards += curSetToParse.getRawRange().cardCount;
|
||||
}
|
||||
cards.reserve(estimatedCards);
|
||||
|
||||
// add an empty set for tokens
|
||||
CardSetPtr tokenSet =
|
||||
|
|
@ -578,11 +573,44 @@ int OracleImporter::startImport()
|
|||
CardSetPtr newSet = CardSet::newInstance(noOpController, curSetToParse.getShortName(),
|
||||
curSetToParse.getLongName(), curSetToParse.getSetType(),
|
||||
curSetToParse.getReleaseDate(), curSetToParse.getPriority());
|
||||
|
||||
// parse only this set's slice of the raw document so the whole JSON tree is
|
||||
// never kept in memory at once
|
||||
const RawJson::SetDataRange &rawRange = curSetToParse.getRawRange();
|
||||
const qsizetype rangeEnd = rawRange.start + rawRange.length;
|
||||
if (rawRange.start < 0 || rawRange.length <= 0 || rangeEnd > rawSetsData.size()) {
|
||||
// rawSetsData is cleared by releaseSetData() while SetToDownload copies
|
||||
// taken from getSets() keep their ranges, and nothing else enforces the
|
||||
// pairing — so never index past the buffer on stale/mismatched ranges.
|
||||
qWarning() << "error: out-of-bounds raw range for set" << curSetToParse.getShortName() << "skipping";
|
||||
++setIndex;
|
||||
emit setIndexChanged(0, setIndex, curSetToParse.getLongName());
|
||||
continue;
|
||||
}
|
||||
// sliced() shares the buffer instead of deep-copying the slice; the largest
|
||||
// sets in AllPrintings are tens of MB, so the copy is worth avoiding here.
|
||||
const QByteArray setBytes = rawSetsData.sliced(rawRange.start, rawRange.length);
|
||||
QJsonParseError parseError;
|
||||
const QJsonDocument setDoc = QJsonDocument::fromJson(setBytes, &parseError);
|
||||
if (parseError.error != QJsonParseError::NoError) {
|
||||
qWarning() << "error: parsing card data for set" << curSetToParse.getShortName() << ":"
|
||||
<< parseError.errorString();
|
||||
++setIndex;
|
||||
// Keep the progress accounting honest: a set that failed to parse
|
||||
// still advanced the index, so report it (with zero imported cards)
|
||||
// rather than letting SaveSetsPage's bar stall per failed set.
|
||||
emit setIndexChanged(0, setIndex, curSetToParse.getLongName());
|
||||
continue;
|
||||
}
|
||||
|
||||
// Only add the set to the database once its slice parsed cleanly;
|
||||
// a set that fails here must not persist as an empty set in cards.xml.
|
||||
if (!sets.contains(newSet->getShortName())) {
|
||||
sets.insert(newSet->getShortName(), newSet);
|
||||
}
|
||||
|
||||
int numCardsInSet = importCardsFromSet(newSet, curSetToParse.getCards());
|
||||
const QJsonArray setCards = setDoc.object().value("cards").toArray();
|
||||
int numCardsInSet = importCardsFromSet(newSet, setCards);
|
||||
|
||||
++setIndex;
|
||||
|
||||
|
|
@ -605,6 +633,7 @@ bool OracleImporter::saveToFile(const QString &fileName, const QString &sourceUr
|
|||
void OracleImporter::releaseSetData()
|
||||
{
|
||||
allSets.clear();
|
||||
rawSetsData.clear();
|
||||
}
|
||||
|
||||
void OracleImporter::clear()
|
||||
|
|
@ -612,4 +641,5 @@ void OracleImporter::clear()
|
|||
sets.clear();
|
||||
cards.clear();
|
||||
allSets.clear();
|
||||
rawSetsData.clear();
|
||||
}
|
||||
|
|
|
|||
|
|
@ -1,6 +1,9 @@
|
|||
#ifndef ORACLEIMPORTER_H
|
||||
#define ORACLEIMPORTER_H
|
||||
|
||||
#include "raw_json_scanner.h"
|
||||
|
||||
#include <QByteArray>
|
||||
#include <QJsonArray>
|
||||
#include <QJsonObject>
|
||||
#include <QMap>
|
||||
|
|
@ -46,10 +49,12 @@ class SetToDownload
|
|||
{
|
||||
private:
|
||||
QString shortName, longName;
|
||||
QJsonArray cards;
|
||||
QDate releaseDate;
|
||||
QString setType;
|
||||
CardSet::Priority priority;
|
||||
// Byte range of this set's object within the importer's raw JSON text. Parsing
|
||||
// one set at a time keeps peak memory low instead of holding the whole document.
|
||||
RawJson::SetDataRange rawRange;
|
||||
|
||||
public:
|
||||
const QString &getShortName() const
|
||||
|
|
@ -60,10 +65,6 @@ public:
|
|||
{
|
||||
return longName;
|
||||
}
|
||||
const QJsonArray &getCards() const
|
||||
{
|
||||
return cards;
|
||||
}
|
||||
const QString &getSetType() const
|
||||
{
|
||||
return setType;
|
||||
|
|
@ -76,16 +77,23 @@ public:
|
|||
{
|
||||
return priority;
|
||||
}
|
||||
const RawJson::SetDataRange &getRawRange() const
|
||||
{
|
||||
return rawRange;
|
||||
}
|
||||
SetToDownload(QString _shortName,
|
||||
QString _longName,
|
||||
QJsonArray _cards,
|
||||
CardSet::Priority _priority,
|
||||
QString _setType = QString(),
|
||||
const QDate &_releaseDate = QDate())
|
||||
: shortName(std::move(_shortName)), longName(std::move(_longName)), cards(std::move(_cards)),
|
||||
releaseDate(_releaseDate), setType(std::move(_setType)), priority(_priority)
|
||||
: shortName(std::move(_shortName)), longName(std::move(_longName)), releaseDate(_releaseDate),
|
||||
setType(std::move(_setType)), priority(_priority)
|
||||
{
|
||||
}
|
||||
void setRawRange(const RawJson::SetDataRange &_rawRange)
|
||||
{
|
||||
rawRange = _rawRange;
|
||||
}
|
||||
bool operator<(const SetToDownload &set) const
|
||||
{
|
||||
return longName.compare(set.longName, Qt::CaseInsensitive) < 0;
|
||||
|
|
@ -141,6 +149,12 @@ private:
|
|||
|
||||
QList<SetToDownload> allSets;
|
||||
|
||||
/**
|
||||
* The raw JSON text of the source document, retained for lazy per-set
|
||||
* parsing during startImport(). Frees the card data as each set is imported.
|
||||
*/
|
||||
QByteArray rawSetsData;
|
||||
|
||||
CardInfoPtr addCard(QString name,
|
||||
const QString &text,
|
||||
bool isToken,
|
||||
|
|
@ -153,7 +167,11 @@ signals:
|
|||
|
||||
public:
|
||||
explicit OracleImporter(QObject *parent = nullptr);
|
||||
bool readSetsFromByteArray(const QByteArray &data);
|
||||
/**
|
||||
* Scans the given JSON document for set metadata. Takes the data by value so
|
||||
* the wizard can hand over its decompressed buffer without copying it.
|
||||
*/
|
||||
bool readSetsFromByteArray(QByteArray data);
|
||||
int startImport();
|
||||
bool saveToFile(const QString &fileName, const QString &sourceUrl, const QString &sourceVersion);
|
||||
int importCardsFromSet(const CardSetPtr ¤tSet, const QJsonArray &cardsList);
|
||||
|
|
@ -169,6 +187,10 @@ public:
|
|||
{
|
||||
return allSets;
|
||||
}
|
||||
const QByteArray &getRawSetsData() const
|
||||
{
|
||||
return rawSetsData;
|
||||
}
|
||||
void releaseSetData();
|
||||
void clear();
|
||||
};
|
||||
|
|
|
|||
621
oracle/src/raw_json_scanner.cpp
Normal file
621
oracle/src/raw_json_scanner.cpp
Normal file
|
|
@ -0,0 +1,621 @@
|
|||
#include "raw_json_scanner.h"
|
||||
|
||||
#include <cstring>
|
||||
|
||||
namespace
|
||||
{
|
||||
|
||||
// Nesting cap matching QJsonDocument's limit, so a pathologically deep document
|
||||
// fails shallowly instead of overflowing the stack through the recursive
|
||||
// skipValue/skipArray/skipObject walk (Qt's parser caps at 1024 for the same
|
||||
// reason and reports DeepNesting).
|
||||
constexpr int kMaxNestingDepth = 1024;
|
||||
|
||||
inline bool isWhitespace(char c)
|
||||
{
|
||||
return c == ' ' || c == '\t' || c == '\r' || c == '\n';
|
||||
}
|
||||
|
||||
const char *skipWhitespace(const char *p, const char *end)
|
||||
{
|
||||
while (p < end && isWhitespace(*p)) {
|
||||
++p;
|
||||
}
|
||||
return p;
|
||||
}
|
||||
|
||||
inline bool isHexDigit(char c)
|
||||
{
|
||||
return (c >= '0' && c <= '9') || (c >= 'a' && c <= 'f') || (c >= 'A' && c <= 'F');
|
||||
}
|
||||
|
||||
inline quint8 hexValue(char c)
|
||||
{
|
||||
if (c >= '0' && c <= '9') {
|
||||
return c - '0';
|
||||
}
|
||||
if (c >= 'a' && c <= 'f') {
|
||||
return c - 'a' + 10;
|
||||
}
|
||||
return c - 'A' + 10;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Skips past a JSON string without decoding it, validating escapes.
|
||||
* @param p In: pointing at the opening quote. Out: pointing past the closing quote.
|
||||
*/
|
||||
bool skipString(const char *&p, const char *end)
|
||||
{
|
||||
++p; // opening quote
|
||||
for (;;) {
|
||||
const void *quote = memchr(p, '"', static_cast<size_t>(end - p));
|
||||
if (!quote) {
|
||||
return false; // unterminated string
|
||||
}
|
||||
// Backslash escapes can only appear before the closing quote, so bound
|
||||
// the scan to the string extent instead of the rest of the document.
|
||||
const void *backslash = memchr(p, '\\', static_cast<size_t>(static_cast<const char *>(quote) - p));
|
||||
if (!backslash) {
|
||||
p = static_cast<const char *>(quote) + 1;
|
||||
return true;
|
||||
}
|
||||
const char *b = static_cast<const char *>(backslash);
|
||||
if (end - b < 2) {
|
||||
return false;
|
||||
}
|
||||
const char escaped = b[1];
|
||||
if (escaped == 'u') {
|
||||
if (end - b < 6) {
|
||||
return false;
|
||||
}
|
||||
quint32 codepoint = 0;
|
||||
for (int i = 0; i < 4; ++i) {
|
||||
if (!isHexDigit(b[2 + i])) {
|
||||
return false;
|
||||
}
|
||||
codepoint = codepoint * 16 + hexValue(b[2 + i]);
|
||||
}
|
||||
p = b + 6;
|
||||
if (codepoint >= 0xD800 && codepoint <= 0xDBFF) {
|
||||
// expect the low-surrogate escape for the second half
|
||||
if (end - p < 6 || p[0] != '\\' || p[1] != 'u') {
|
||||
return false; // unpaired high surrogate
|
||||
}
|
||||
quint32 low = 0;
|
||||
for (int i = 0; i < 4; ++i) {
|
||||
if (!isHexDigit(p[2 + i])) {
|
||||
return false;
|
||||
}
|
||||
low = low * 16 + hexValue(p[2 + i]);
|
||||
}
|
||||
if (low < 0xDC00 || low > 0xDFFF) {
|
||||
return false;
|
||||
}
|
||||
p += 6;
|
||||
} else if (codepoint >= 0xDC00 && codepoint <= 0xDFFF) {
|
||||
return false; // unpaired low surrogate
|
||||
}
|
||||
continue;
|
||||
}
|
||||
switch (escaped) {
|
||||
case '"':
|
||||
case '\\':
|
||||
case '/':
|
||||
case 'b':
|
||||
case 'f':
|
||||
case 'n':
|
||||
case 'r':
|
||||
case 't':
|
||||
p = b + 2;
|
||||
continue;
|
||||
default:
|
||||
return false; // invalid escape
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Decodes a JSON string into @p out, validating it as it goes.
|
||||
* @param p In: pointing at the opening quote. Out: pointing past the closing quote.
|
||||
*/
|
||||
bool decodeString(const char *&p, const char *end, QString &out)
|
||||
{
|
||||
out.clear();
|
||||
QByteArray utf8;
|
||||
auto flush = [&out, &utf8]() {
|
||||
if (!utf8.isEmpty()) {
|
||||
out += QString::fromUtf8(utf8);
|
||||
utf8.clear();
|
||||
}
|
||||
};
|
||||
|
||||
++p; // opening quote
|
||||
while (p < end) {
|
||||
const char c = *p;
|
||||
if (c == '\\') {
|
||||
flush();
|
||||
++p; // escaped character
|
||||
if (p >= end) {
|
||||
return false;
|
||||
}
|
||||
const char escaped = *p;
|
||||
if (escaped == 'u') {
|
||||
++p; // first hex digit
|
||||
if (p + 4 > end) {
|
||||
return false;
|
||||
}
|
||||
quint32 codepoint = 0;
|
||||
for (int i = 0; i < 4; ++i) {
|
||||
if (!isHexDigit(p[i])) {
|
||||
return false;
|
||||
}
|
||||
codepoint = codepoint * 16 + hexValue(p[i]);
|
||||
}
|
||||
p += 4;
|
||||
if (codepoint >= 0xD800 && codepoint <= 0xDBFF) {
|
||||
// expect a low-surrogate escape for the second half
|
||||
if (p + 6 > end || p[0] != '\\' || p[1] != 'u') {
|
||||
return false; // unpaired high surrogate
|
||||
}
|
||||
quint32 low = 0;
|
||||
for (int i = 0; i < 4; ++i) {
|
||||
if (!isHexDigit(p[2 + i])) {
|
||||
return false;
|
||||
}
|
||||
low = low * 16 + hexValue(p[2 + i]);
|
||||
}
|
||||
if (low < 0xDC00 || low > 0xDFFF) {
|
||||
return false;
|
||||
}
|
||||
out += QChar(codepoint);
|
||||
out += QChar(low);
|
||||
p += 6;
|
||||
} else if (codepoint >= 0xDC00 && codepoint <= 0xDFFF) {
|
||||
return false; // unpaired low surrogate
|
||||
} else {
|
||||
out += QChar(codepoint);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
switch (escaped) {
|
||||
case '"':
|
||||
out += '"';
|
||||
break;
|
||||
case '\\':
|
||||
out += '\\';
|
||||
break;
|
||||
case '/':
|
||||
out += '/';
|
||||
break;
|
||||
case 'b':
|
||||
out += '\b';
|
||||
break;
|
||||
case 'f':
|
||||
out += '\f';
|
||||
break;
|
||||
case 'n':
|
||||
out += '\n';
|
||||
break;
|
||||
case 'r':
|
||||
out += '\r';
|
||||
break;
|
||||
case 't':
|
||||
out += '\t';
|
||||
break;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
++p;
|
||||
continue;
|
||||
}
|
||||
if (c == '"') {
|
||||
++p;
|
||||
flush();
|
||||
return true;
|
||||
}
|
||||
// Deliberately accept unescaped control characters (e.g. a tab inside
|
||||
// a set name): QJsonDocument and skipString accept them too, so
|
||||
// rejecting them here would fail the whole document on a byte that
|
||||
// Qt is fine with — the very total-failure mode this scanner avoids.
|
||||
utf8 += c;
|
||||
++p;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Reads a set-metadata field, tolerating null and non-string values.
|
||||
*
|
||||
* A set's metadata may carry null or non-string values in otherwise-valid
|
||||
* payloads ("releaseDate": null, "type": 7). The token itself was already
|
||||
* structurally validated by skipValue, so a non-string value is accepted and
|
||||
* leaves @p out at its default (empty) — one bad set must not abort the
|
||||
* import of every other set in the document.
|
||||
*/
|
||||
bool decodeStringMember(const char *&fs, const char *&fe, QString &out)
|
||||
{
|
||||
if (fs >= fe) {
|
||||
return false;
|
||||
}
|
||||
if (*fs != '"') {
|
||||
return true;
|
||||
}
|
||||
return decodeString(fs, fe, out);
|
||||
}
|
||||
|
||||
bool matchLiteral(const char *&p, const char *end, const char *literal, int length)
|
||||
{
|
||||
if (end - p < length || memcmp(p, literal, static_cast<size_t>(length)) != 0) {
|
||||
return false;
|
||||
}
|
||||
const char *after = p + length;
|
||||
if (after < end && (QChar::isLetter(*after) || QChar::isDigit(*after) || *after == '_')) {
|
||||
return false;
|
||||
}
|
||||
p = after;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool skipNumber(const char *&p, const char *end)
|
||||
{
|
||||
// JSON number: -?(0|[1-9][0-9]*)(\.[0-9]+)?([eE][+-]?[0-9]+)?
|
||||
if (p < end && *p == '-') {
|
||||
++p;
|
||||
}
|
||||
if (p < end && *p == '0') {
|
||||
++p;
|
||||
} else if (p < end && *p >= '1' && *p <= '9') {
|
||||
++p;
|
||||
while (p < end && QChar::isDigit(*p)) {
|
||||
++p;
|
||||
}
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
if (p < end && *p == '.') {
|
||||
++p;
|
||||
if (p >= end || !QChar::isDigit(*p)) {
|
||||
return false;
|
||||
}
|
||||
while (p < end && QChar::isDigit(*p)) {
|
||||
++p;
|
||||
}
|
||||
}
|
||||
if (p < end && (*p == 'e' || *p == 'E')) {
|
||||
++p;
|
||||
if (p < end && (*p == '+' || *p == '-')) {
|
||||
++p;
|
||||
}
|
||||
if (p >= end || !QChar::isDigit(*p)) {
|
||||
return false;
|
||||
}
|
||||
while (p < end && QChar::isDigit(*p)) {
|
||||
++p;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool skipValue(const char *&p, const char *end, int depth);
|
||||
bool skipObject(const char *&p, const char *end, int depth);
|
||||
bool skipArray(const char *&p, const char *end, int depth);
|
||||
|
||||
bool skipPrimitive(const char *&p, const char *end)
|
||||
{
|
||||
if (p >= end) {
|
||||
return false;
|
||||
}
|
||||
const char c = *p;
|
||||
if (c == '"') {
|
||||
return skipString(p, end);
|
||||
}
|
||||
if (c == 't') {
|
||||
return matchLiteral(p, end, "true", 4);
|
||||
}
|
||||
if (c == 'f') {
|
||||
return matchLiteral(p, end, "false", 5);
|
||||
}
|
||||
if (c == 'n') {
|
||||
return matchLiteral(p, end, "null", 4);
|
||||
}
|
||||
if (c == '-' || (c >= '0' && c <= '9')) {
|
||||
return skipNumber(p, end);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
bool skipObject(const char *&p, const char *end, int depth)
|
||||
{
|
||||
if (depth <= 0) {
|
||||
return false; // nest deeper than the cap
|
||||
}
|
||||
++p; // '{'
|
||||
p = skipWhitespace(p, end);
|
||||
if (p < end && *p == '}') {
|
||||
++p;
|
||||
return true;
|
||||
}
|
||||
for (;;) {
|
||||
p = skipWhitespace(p, end);
|
||||
if (p >= end || *p != '"') {
|
||||
return false;
|
||||
}
|
||||
if (!skipString(p, end)) {
|
||||
return false;
|
||||
}
|
||||
p = skipWhitespace(p, end);
|
||||
if (p >= end || *p != ':') {
|
||||
return false;
|
||||
}
|
||||
++p;
|
||||
if (!skipValue(p, end, depth - 1)) {
|
||||
return false;
|
||||
}
|
||||
p = skipWhitespace(p, end);
|
||||
if (p >= end) {
|
||||
return false;
|
||||
}
|
||||
if (*p == ',') {
|
||||
++p;
|
||||
continue;
|
||||
}
|
||||
if (*p == '}') {
|
||||
++p;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
bool skipArray(const char *&p, const char *end, int depth)
|
||||
{
|
||||
if (depth <= 0) {
|
||||
return false; // nest deeper than the cap
|
||||
}
|
||||
++p; // '['
|
||||
p = skipWhitespace(p, end);
|
||||
if (p < end && *p == ']') {
|
||||
++p;
|
||||
return true;
|
||||
}
|
||||
for (;;) {
|
||||
if (!skipValue(p, end, depth - 1)) {
|
||||
return false;
|
||||
}
|
||||
p = skipWhitespace(p, end);
|
||||
if (p >= end) {
|
||||
return false;
|
||||
}
|
||||
if (*p == ',') {
|
||||
++p;
|
||||
continue;
|
||||
}
|
||||
if (*p == ']') {
|
||||
++p;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
bool skipValue(const char *&p, const char *end, int depth)
|
||||
{
|
||||
p = skipWhitespace(p, end);
|
||||
if (p >= end) {
|
||||
return false;
|
||||
}
|
||||
const char c = *p;
|
||||
if (c == '{') {
|
||||
// pass depth through: skipObject consumes the single decrement for this level
|
||||
return skipObject(p, end, depth);
|
||||
}
|
||||
if (c == '[') {
|
||||
return skipArray(p, end, depth);
|
||||
}
|
||||
// a primitive is a leaf, so it never wastes a nesting level
|
||||
return skipPrimitive(p, end);
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Iterates the members of the object starting at @p p.
|
||||
*
|
||||
* For each member invokes @p memberCallback with the key and the byte range of
|
||||
* its value. Advancing @p p is unaffected by the callback.
|
||||
*/
|
||||
template <typename F> bool forEachObjectMember(const char *&p, const char *end, int depth, F &&memberCallback)
|
||||
{
|
||||
if (depth <= 0) {
|
||||
return false; // nest deeper than the cap
|
||||
}
|
||||
++p; // '{'
|
||||
p = skipWhitespace(p, end);
|
||||
if (p < end && *p == '}') {
|
||||
++p;
|
||||
return true;
|
||||
}
|
||||
for (;;) {
|
||||
p = skipWhitespace(p, end);
|
||||
if (p >= end || *p != '"') {
|
||||
return false;
|
||||
}
|
||||
QString key;
|
||||
if (!decodeString(p, end, key)) {
|
||||
return false;
|
||||
}
|
||||
p = skipWhitespace(p, end);
|
||||
if (p >= end || *p != ':') {
|
||||
return false;
|
||||
}
|
||||
++p;
|
||||
const char *valueStart = skipWhitespace(p, end);
|
||||
const char *valueEnd = valueStart;
|
||||
if (!skipValue(valueEnd, end, depth - 1)) {
|
||||
return false;
|
||||
}
|
||||
if (!memberCallback(key, valueStart, valueEnd)) {
|
||||
return false;
|
||||
}
|
||||
p = valueEnd;
|
||||
p = skipWhitespace(p, end);
|
||||
if (p >= end) {
|
||||
return false;
|
||||
}
|
||||
if (*p == ',') {
|
||||
++p;
|
||||
continue;
|
||||
}
|
||||
if (*p == '}') {
|
||||
++p;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// Counts the direct elements of an array value; returns -1 if the array is malformed.
|
||||
int countArrayElements(const char *p, const char *end, int depth)
|
||||
{
|
||||
if (depth <= 0) {
|
||||
return -1; // nest deeper than the cap
|
||||
}
|
||||
++p; // '['
|
||||
p = skipWhitespace(p, end);
|
||||
int count = 0;
|
||||
if (p < end && *p == ']') {
|
||||
return 0;
|
||||
}
|
||||
for (;;) {
|
||||
if (!skipValue(p, end, depth - 1)) {
|
||||
return -1;
|
||||
}
|
||||
++count;
|
||||
p = skipWhitespace(p, end);
|
||||
if (p >= end) {
|
||||
return -1;
|
||||
}
|
||||
if (*p == ',') {
|
||||
++p;
|
||||
continue;
|
||||
}
|
||||
if (*p == ']') {
|
||||
return count;
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
namespace RawJson
|
||||
{
|
||||
|
||||
QList<SetRange> scanSetRanges(const QByteArray &json, ScanError *error)
|
||||
{
|
||||
QList<SetRange> ranges;
|
||||
if (error) {
|
||||
*error = ScanError{};
|
||||
}
|
||||
|
||||
const auto fail = [&](const QString &message) -> QList<SetRange> {
|
||||
if (error) {
|
||||
error->message = message;
|
||||
}
|
||||
return {};
|
||||
};
|
||||
|
||||
const char *begin = json.constData();
|
||||
const char *end = begin + json.size();
|
||||
if (begin >= end) {
|
||||
return fail(QStringLiteral("empty JSON document"));
|
||||
}
|
||||
|
||||
const char *p = skipWhitespace(begin, end);
|
||||
if (p >= end || *p != '{') {
|
||||
return fail(QStringLiteral("top-level JSON must be an object"));
|
||||
}
|
||||
|
||||
bool foundData = false;
|
||||
bool malformedSetData = false;
|
||||
|
||||
const auto topLevelCallback = [&](const QString &key, const char *valueStart, const char *valueEnd) {
|
||||
if (key == QStringLiteral("data")) {
|
||||
foundData = true;
|
||||
if (valueStart >= valueEnd || *valueStart != '{') {
|
||||
malformedSetData = true;
|
||||
return false;
|
||||
}
|
||||
const char *setP = valueStart;
|
||||
const bool ok = forEachObjectMember(setP, valueEnd, kMaxNestingDepth - 1,
|
||||
[&](const QString &setCode, const char *setStart, const char *setEnd) {
|
||||
if (setStart >= setEnd || *setStart != '{') {
|
||||
malformedSetData = true;
|
||||
return false;
|
||||
}
|
||||
SetRange range;
|
||||
range.dataRange.start = setStart - begin;
|
||||
range.dataRange.length = setEnd - setStart;
|
||||
range.code = setCode;
|
||||
|
||||
const char *memberP = setStart;
|
||||
const bool metaOk = forEachObjectMember(
|
||||
memberP, setEnd, kMaxNestingDepth - 2,
|
||||
[&](const QString &field, const char *fs, const char *fe) {
|
||||
if (field == QStringLiteral("code")) {
|
||||
return decodeStringMember(fs, fe, range.code);
|
||||
}
|
||||
if (field == QStringLiteral("name")) {
|
||||
return decodeStringMember(fs, fe, range.name);
|
||||
}
|
||||
if (field == QStringLiteral("type")) {
|
||||
return decodeStringMember(fs, fe, range.type);
|
||||
}
|
||||
if (field == QStringLiteral("releaseDate")) {
|
||||
return decodeStringMember(fs, fe, range.releaseDate);
|
||||
}
|
||||
if (field == QStringLiteral("cards")) {
|
||||
if (fs >= fe) {
|
||||
return false;
|
||||
}
|
||||
if (*fs != '[') {
|
||||
// e.g. "cards": null — treat as an empty array,
|
||||
// matching Qt's tolerance.
|
||||
return true;
|
||||
}
|
||||
range.dataRange.cardCount =
|
||||
countArrayElements(fs, fe, kMaxNestingDepth - 2);
|
||||
return range.dataRange.cardCount >= 0;
|
||||
}
|
||||
return true;
|
||||
});
|
||||
if (!metaOk) {
|
||||
malformedSetData = true;
|
||||
return false;
|
||||
}
|
||||
ranges.append(range);
|
||||
return true;
|
||||
});
|
||||
if (!ok) {
|
||||
malformedSetData = true;
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
};
|
||||
|
||||
if (!forEachObjectMember(p, end, kMaxNestingDepth, topLevelCallback)) {
|
||||
return fail(malformedSetData ? QStringLiteral("malformed set data") : QStringLiteral("malformed JSON"));
|
||||
}
|
||||
p = skipWhitespace(p, end);
|
||||
if (p != end) {
|
||||
return fail(QStringLiteral("trailing content after top-level JSON object"));
|
||||
}
|
||||
if (!foundData) {
|
||||
return fail(QStringLiteral("missing \"data\" object"));
|
||||
}
|
||||
if (ranges.isEmpty()) {
|
||||
return fail(QStringLiteral("no sets found in \"data\""));
|
||||
}
|
||||
return ranges;
|
||||
}
|
||||
|
||||
} // namespace RawJson
|
||||
76
oracle/src/raw_json_scanner.h
Normal file
76
oracle/src/raw_json_scanner.h
Normal file
|
|
@ -0,0 +1,76 @@
|
|||
#ifndef RAW_JSON_SCANNER_H
|
||||
#define RAW_JSON_SCANNER_H
|
||||
|
||||
#include <QByteArray>
|
||||
#include <QList>
|
||||
#include <QString>
|
||||
|
||||
namespace RawJson
|
||||
{
|
||||
|
||||
/**
|
||||
* @brief The byte extent of a set's object inside the scanned document, plus
|
||||
* the size of its cards array. This is the slice SetToDownload needs for lazy
|
||||
* per-set parsing; the metadata strings live in SetRange alongside it.
|
||||
*/
|
||||
struct SetDataRange
|
||||
{
|
||||
/** @brief Byte offset of the set's object within the scanned buffer. */
|
||||
qsizetype start = -1;
|
||||
/** @brief Byte length of the set's object, including the surrounding braces. */
|
||||
qsizetype length = 0;
|
||||
/** @brief Number of entries in the set's "cards" array. */
|
||||
int cardCount = 0;
|
||||
};
|
||||
|
||||
struct SetRange
|
||||
{
|
||||
/** @brief The byte slice of this set within the document. */
|
||||
SetDataRange dataRange;
|
||||
QString code;
|
||||
QString name;
|
||||
QString type;
|
||||
QString releaseDate;
|
||||
};
|
||||
|
||||
struct ScanError
|
||||
{
|
||||
bool isError() const
|
||||
{
|
||||
return !message.isEmpty();
|
||||
}
|
||||
QString message;
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Scans a full MTGJSON document without materializing the JSON tree.
|
||||
*
|
||||
* Splits the top-level "data" object into per-set byte ranges and reads each
|
||||
* set's metadata directly from the raw bytes. The oracle importer can then
|
||||
* parse one set at a time during import, keeping peak memory far below a single
|
||||
* QJsonDocument::fromJson() over the whole file.
|
||||
*
|
||||
* The whole document is structurally validated while scanning (strings,
|
||||
* escapes, braces, and a trailing-content check) and nesting depth is capped at
|
||||
* 1024 to match QJsonDocument, so pathologically deep documents fail shallowly
|
||||
* instead of exhausting the stack. Verdicts agree with QJsonDocument::fromJson
|
||||
* on structurally malformed input; unlike Qt, string metadata fields
|
||||
* ("name", "type", "releaseDate", "code") tolerate null / non-string values by
|
||||
* defaulting to empty rather than rejecting the whole document, so one broken
|
||||
* set cannot abort the import of the rest.
|
||||
*
|
||||
* Following QJsonDocument::fromJson's convention, the parsed ranges are
|
||||
* returned by value and any failure is reported through the @p error out
|
||||
* parameter.
|
||||
*
|
||||
* @param json The raw MTGJSON document bytes.
|
||||
* @param error Out parameter. Set to an error ScanError when the document
|
||||
* cannot be parsed, otherwise left empty. Passing a null
|
||||
* pointer disables error reporting.
|
||||
* @return The detected per-set ranges, or an empty list on failure.
|
||||
*/
|
||||
QList<SetRange> scanSetRanges(const QByteArray &json, ScanError *error = nullptr);
|
||||
|
||||
} // namespace RawJson
|
||||
|
||||
#endif // RAW_JSON_SCANNER_H
|
||||
Loading…
Add table
Add a link
Reference in a new issue