ViewVC Help
View File | Revision Log | Show Annotations | Download File
/cvs/cvsroot/cf.schmorp.de/dclient/src/backend/json/json_reader.cpp
Revision: 1.2
Committed: Mon Oct 18 16:38:02 2010 UTC (15 years, 11 months ago) by sf-pippijn
Branch: MAIN
CVS Tags: HEAD
Changes since 1.1: +2 -2 lines
Log Message:
support msg colours

File Contents

# User Rev Content
1 sf-pippijn 1.1 #include <json/reader.h>
2     #include <json/value.h>
3    
4     #include <cassert>
5     #include <cstdio>
6     #include <cstring>
7     #include <iostream>
8     #include <stdexcept>
9     #include <utility>
10    
11     #if _MSC_VER >= 1400 /* VC++ 8.0 */
12     #pragma warning( disable : 4996 ) /* disable warning about strdup being deprecated. */
13     #endif
14    
15     namespace Json
16     {
17    
18     /*
19     * Implementation of class Features
20     * ////////////////////////////////
21     */
22    
23     Features::Features ()
24     : allowComments_ (true)
25     , strictRoot_ (false)
26     {
27     }
28    
29     Features
30     Features::all ()
31     {
32     return Features ();
33     }
34    
35     Features
36     Features::strictMode ()
37     {
38     Features features;
39    
40     features.allowComments_ = false;
41     features.strictRoot_ = true;
42     return features;
43     }
44    
45     /*
46     * Implementation of class Reader
47     * ////////////////////////////////
48     */
49    
50    
51     static inline bool
52     in (Reader::Char c, Reader::Char c1, Reader::Char c2, Reader::Char c3, Reader::Char c4)
53     {
54     return c == c1 || c == c2 || c == c3 || c == c4;
55     }
56    
57     static inline bool
58     in (Reader::Char c, Reader::Char c1, Reader::Char c2, Reader::Char c3, Reader::Char c4, Reader::Char c5)
59     {
60     return c == c1 || c == c2 || c == c3 || c == c4 || c == c5;
61     }
62    
63     static bool
64     containsNewLine (Reader::Location begin, Reader::Location end)
65     {
66     for (; begin < end; ++begin)
67     if (*begin == '\n' || *begin == '\r')
68     return true;
69     return false;
70     }
71    
72     static std::string
73     codePointToUTF8 (unsigned int cp)
74     {
75     std::string result;
76    
77     /* based on description from http://en.wikipedia.org/wiki/UTF-8 */
78    
79     if (cp <= 0x7f)
80     {
81     result.resize (1);
82     result[0] = static_cast<char> (cp);
83     }
84     else if (cp <= 0x7FF)
85     {
86     result.resize (2);
87     result[1] = static_cast<char> (0x80 | (0x3f & cp));
88     result[0] = static_cast<char> (0xC0 | (0x1f & (cp >> 6)));
89     }
90     else if (cp <= 0xFFFF)
91     {
92     result.resize (3);
93     result[2] = static_cast<char> (0x80 | (0x3f & cp));
94     result[1] = 0x80 | static_cast<char> ((0x3f & (cp >> 6)));
95     result[0] = 0xE0 | static_cast<char> ((0xf & (cp >> 12)));
96     }
97     else if (cp <= 0x10FFFF)
98     {
99     result.resize (4);
100     result[3] = static_cast<char> (0x80 | (0x3f & cp));
101     result[2] = static_cast<char> (0x80 | (0x3f & (cp >> 6)));
102     result[1] = static_cast<char> (0x80 | (0x3f & (cp >> 12)));
103     result[0] = static_cast<char> (0xF0 | (0x7 & (cp >> 18)));
104     }
105    
106     return result;
107     }
108    
109     /*
110     * Class Reader
111     * //////////////////////////////////////////////////////////////////
112     */
113    
114     Reader::Reader ()
115     : features_ (Features::all ())
116     {
117     }
118    
119     Reader::Reader (const Features &features)
120     : features_ (features)
121     {
122     }
123    
124     bool
125     Reader::parse (const std::string &document, Value &root, bool collectComments)
126     {
127     document_ = document;
128     const char *begin = document_.c_str ();
129     const char *end = begin + document_.length ();
130     return parse (begin, end, root, collectComments);
131     }
132    
133     bool
134     Reader::parse (std::istream &sin, Value &root, bool collectComments)
135     {
136     /*
137     * std::istream_iterator<char> begin(sin);
138     * std::istream_iterator<char> end;
139     * Those would allow streamed input from a file, if parse() were a
140     * template function.
141     */
142    
143     /*
144     * Since std::string is reference-counted, this at least does not
145     * create an extra copy.
146     */
147     std::string doc;
148    
149     std::getline (sin, doc, (char)EOF);
150     return parse (doc, root, collectComments);
151     }
152    
153     bool
154     Reader::parse (const char *beginDoc, const char *endDoc, Value &root, bool collectComments)
155     {
156     if (!features_.allowComments_)
157     collectComments = false;
158    
159     begin_ = beginDoc;
160     end_ = endDoc;
161     collectComments_ = collectComments;
162     current_ = begin_;
163     lastValueEnd_ = 0;
164     lastValue_ = 0;
165     commentsBefore_ = "";
166     errors_.clear ();
167     while (!nodes_.empty ())
168     nodes_.pop ();
169     nodes_.push (&root);
170    
171     bool successful = readValue ();
172     Token token;
173     skipCommentTokens (token);
174     if (collectComments_ && !commentsBefore_.empty ())
175     root.setComment (commentsBefore_, commentAfter);
176     if (features_.strictRoot_)
177     if (!root.isArray () && !root.isObject ())
178     {
179     /* Set error location to start of doc, ideally should be first token found in doc */
180     token.type_ = tokenError;
181     token.start_ = beginDoc;
182     token.end_ = endDoc;
183     addError ("A valid JSON document must be either an array or an object value.",
184     token);
185     return false;
186     }
187     return successful;
188     }
189    
190     bool
191     Reader::readValue ()
192     {
193     Token token;
194    
195     skipCommentTokens (token);
196     bool successful = true;
197    
198     if (collectComments_ && !commentsBefore_.empty ())
199     {
200     currentValue ().setComment (commentsBefore_, commentBefore);
201     commentsBefore_ = "";
202     }
203    
204    
205     switch (token.type_)
206     {
207     case tokenObjectBegin:
208     successful = readObject (token);
209     break;
210     case tokenArrayBegin:
211     successful = readArray (token);
212     break;
213     case tokenNumber:
214     successful = decodeNumber (token);
215     break;
216     case tokenString:
217     successful = decodeString (token);
218     break;
219     case tokenTrue:
220     currentValue () = true;
221     break;
222     case tokenFalse:
223     currentValue () = false;
224     break;
225     case tokenNull:
226     currentValue () = Value ();
227     break;
228     default:
229     return addError ("Syntax error: value, object or array expected.", token);
230     }
231    
232     if (collectComments_)
233     {
234     lastValueEnd_ = current_;
235     lastValue_ = &currentValue ();
236     }
237    
238     return successful;
239     }
240    
241     void
242     Reader::skipCommentTokens (Token &token)
243     {
244     if (features_.allowComments_)
245     {
246     do
247     readToken (token);
248     while (token.type_ == tokenComment);
249     }
250     else
251     readToken (token);
252     }
253    
254     bool
255     Reader::expectToken (TokenType type, Token &token, const char *message)
256     {
257     readToken (token);
258     if (token.type_ != type)
259     return addError (message, token);
260     return true;
261     }
262    
263     bool
264     Reader::readToken (Token &token)
265     {
266     skipSpaces ();
267     token.start_ = current_;
268     Char c = getNextChar ();
269     bool ok = true;
270     switch (c)
271     {
272     case '{':
273     token.type_ = tokenObjectBegin;
274     break;
275     case '}':
276     token.type_ = tokenObjectEnd;
277     break;
278     case '[':
279     token.type_ = tokenArrayBegin;
280     break;
281     case ']':
282     token.type_ = tokenArrayEnd;
283     break;
284     case '"':
285     token.type_ = tokenString;
286     ok = readString ();
287     break;
288     case '/':
289     token.type_ = tokenComment;
290     ok = readComment ();
291     break;
292     case '0':
293     case '1':
294     case '2':
295     case '3':
296     case '4':
297     case '5':
298     case '6':
299     case '7':
300     case '8':
301     case '9':
302     case '-':
303     token.type_ = tokenNumber;
304     readNumber ();
305     break;
306     case 't':
307     token.type_ = tokenTrue;
308     ok = match ("rue", 3);
309     break;
310     case 'f':
311     token.type_ = tokenFalse;
312     ok = match ("alse", 4);
313     break;
314     case 'n':
315     token.type_ = tokenNull;
316     ok = match ("ull", 3);
317     break;
318     case ',':
319     token.type_ = tokenArraySeparator;
320     break;
321     case ':':
322     token.type_ = tokenMemberSeparator;
323     break;
324     case 0:
325     token.type_ = tokenEndOfStream;
326     break;
327     default:
328     ok = false;
329     break;
330     }
331     if (!ok)
332     token.type_ = tokenError;
333     token.end_ = current_;
334     return true;
335     }
336    
337     void
338     Reader::skipSpaces ()
339     {
340     while (current_ != end_)
341     {
342     Char c = *current_;
343     if (c == ' ' || c == '\t' || c == '\r' || c == '\n')
344     ++current_;
345     else
346     break;
347     }
348     }
349    
350     bool
351     Reader::match (Location pattern, int patternLength)
352     {
353     if (end_ - current_ < patternLength)
354     return false;
355     int index = patternLength;
356     while (index--)
357     if (current_[index] != pattern[index])
358     return false;
359     current_ += patternLength;
360     return true;
361     }
362    
363     bool
364     Reader::readComment ()
365     {
366     Location commentBegin = current_ - 1;
367     Char c = getNextChar ();
368     bool successful = false;
369    
370     if (c == '*')
371     successful = readCStyleComment ();
372     else if (c == '/')
373     successful = readCppStyleComment ();
374     if (!successful)
375     return false;
376    
377     if (collectComments_)
378     {
379     CommentPlacement placement = commentBefore;
380     if (lastValueEnd_ && !containsNewLine (lastValueEnd_, commentBegin))
381     if (c != '*' || !containsNewLine (commentBegin, current_))
382     placement = commentAfterOnSameLine;
383    
384     addComment (commentBegin, current_, placement);
385     }
386     return true;
387     }
388    
389     void
390     Reader::addComment (Location begin, Location end, CommentPlacement placement)
391     {
392     assert (collectComments_);
393     if (placement == commentAfterOnSameLine)
394     {
395     assert (lastValue_ != 0);
396     lastValue_->setComment (std::string (begin, end), placement);
397     }
398     else
399     {
400     if (!commentsBefore_.empty ())
401     commentsBefore_ += "\n";
402     commentsBefore_ += std::string (begin, end);
403     }
404     }
405    
406     bool
407     Reader::readCStyleComment ()
408     {
409     while (current_ != end_)
410     {
411     Char c = getNextChar ();
412     if (c == '*' && *current_ == '/')
413     break;
414     }
415     return getNextChar () == '/';
416     }
417    
418     bool
419     Reader::readCppStyleComment ()
420     {
421     while (current_ != end_)
422     {
423     Char c = getNextChar ();
424     if (c == '\r' || c == '\n')
425     break;
426     }
427     return true;
428     }
429    
430     void
431     Reader::readNumber ()
432     {
433     while (current_ != end_)
434     {
435     if (!(*current_ >= '0' && *current_ <= '9') &&
436     !in (*current_, '.', 'e', 'E', '+', '-'))
437     break;
438     ++current_;
439     }
440     }
441    
442     bool
443     Reader::readString ()
444     {
445     Char c = 0;
446    
447     while (current_ != end_)
448     {
449     c = getNextChar ();
450     if (c == '\\')
451     getNextChar ();
452     else if (c == '"')
453     break;
454     }
455     return c == '"';
456     }
457    
458     bool
459     Reader::readObject (Token &tokenStart)
460     {
461     Token tokenName;
462     std::string name;
463    
464     currentValue () = Value (objectValue);
465     while (readToken (tokenName))
466     {
467     bool initialTokenOk = true;
468     while (tokenName.type_ == tokenComment && initialTokenOk)
469     initialTokenOk = readToken (tokenName);
470     if (!initialTokenOk)
471     break;
472     if (tokenName.type_ == tokenObjectEnd && name.empty ()) /* empty object */
473     return true;
474     if (tokenName.type_ != tokenString)
475     break;
476    
477     name = "";
478     if (!decodeString (tokenName, name))
479     return recoverFromError (tokenObjectEnd);
480    
481     Token colon;
482     if (!readToken (colon) || colon.type_ != tokenMemberSeparator)
483     return addErrorAndRecover ("Missing ':' after object member name",
484     colon,
485     tokenObjectEnd);
486     Value &value = currentValue ()[name];
487     nodes_.push (&value);
488     bool ok = readValue ();
489     nodes_.pop ();
490     if (!ok) /* error already set */
491     return recoverFromError (tokenObjectEnd);
492    
493     Token comma;
494     if (!readToken (comma)
495     || (comma.type_ != tokenObjectEnd &&
496     comma.type_ != tokenArraySeparator &&
497     comma.type_ != tokenComment))
498     return addErrorAndRecover ("Missing ',' or '}' in object declaration",
499     comma,
500     tokenObjectEnd);
501     bool finalizeTokenOk = true;
502     while (comma.type_ == tokenComment &&
503     finalizeTokenOk)
504     finalizeTokenOk = readToken (comma);
505     if (comma.type_ == tokenObjectEnd)
506     return true;
507     }
508     return addErrorAndRecover ("Missing '}' or object member name",
509     tokenName,
510     tokenObjectEnd);
511     }
512    
513     bool
514     Reader::readArray (Token &tokenStart)
515     {
516     currentValue () = Value (arrayValue);
517     skipSpaces ();
518     if (*current_ == ']') /* empty array */
519     {
520     Token endArray;
521     readToken (endArray);
522     return true;
523     }
524     int index = 0;
525     while (true)
526     {
527     Value &value = currentValue ()[index++];
528     nodes_.push (&value);
529     bool ok = readValue ();
530     nodes_.pop ();
531     if (!ok) /* error already set */
532     return recoverFromError (tokenArrayEnd);
533    
534     Token token;
535     /* Accept Comment after last item in the array. */
536     ok = readToken (token);
537     while (token.type_ == tokenComment && ok)
538     ok = readToken (token);
539     bool badTokenType = (token.type_ == tokenArraySeparator &&
540     token.type_ == tokenArrayEnd);
541     if (!ok || badTokenType)
542     return addErrorAndRecover ("Missing ',' or ']' in array declaration",
543     token,
544     tokenArrayEnd);
545     if (token.type_ == tokenArrayEnd)
546     break;
547     }
548     return true;
549     }
550    
551     bool
552     Reader::decodeNumber (Token &token)
553     {
554     bool isDouble = false;
555    
556     for (Location inspect = token.start_; inspect != token.end_; ++inspect)
557     isDouble = isDouble
558     || in (*inspect, '.', 'e', 'E', '+')
559     || (*inspect == '-' && inspect != token.start_);
560     if (isDouble)
561     return decodeDouble (token);
562     Location current = token.start_;
563     bool isNegative = *current == '-';
564     if (isNegative)
565     ++current;
566     Value::UInt threshold = (isNegative ? Value::UInt (-Value::minInt)
567     : Value::maxUInt) / 10;
568     Value::UInt value = 0;
569     while (current < token.end_)
570     {
571     Char c = *current++;
572     if (c < '0' || c > '9')
573     return addError ("'" + std::string (token.start_, token.end_) + "' is not a number.", token);
574     if (value >= threshold)
575     return decodeDouble (token);
576     value = value * 10 + Value::UInt (c - '0');
577     }
578     if (isNegative)
579     currentValue () = -Value::Int (value);
580     else if (value <= Value::UInt (Value::maxInt))
581     currentValue () = Value::Int (value);
582     else
583     currentValue () = value;
584     return true;
585     }
586    
587     bool
588     Reader::decodeDouble (Token &token)
589     {
590     double value = 0;
591     const int bufferSize = 32;
592     int count;
593     int length = int(token.end_ - token.start_);
594    
595     if (length <= bufferSize)
596     {
597     Char buffer[bufferSize];
598     memcpy (buffer, token.start_, length);
599     buffer[length] = 0;
600     count = sscanf (buffer, "%lf", &value);
601     }
602     else
603     {
604     std::string buffer (token.start_, token.end_);
605     count = sscanf (buffer.c_str (), "%lf", &value);
606     }
607    
608     if (count != 1)
609     return addError ("'" + std::string (token.start_, token.end_) + "' is not a number.", token);
610     currentValue () = value;
611     return true;
612     }
613    
614     bool
615     Reader::decodeString (Token &token)
616     {
617     std::string decoded;
618    
619     if (!decodeString (token, decoded))
620     return false;
621     currentValue () = decoded;
622     return true;
623     }
624    
625     bool
626     Reader::decodeString (Token &token, std::string &decoded)
627     {
628     decoded.reserve (token.end_ - token.start_ - 2);
629     Location current = token.start_ + 1; /* skip '"' */
630     Location end = token.end_ - 1; /* do not include '"' */
631     while (current != end)
632     {
633     Char c = *current++;
634     if (c == '"')
635     break;
636     else if (c == '\\')
637     {
638     if (current == end)
639     return addError ("Empty escape sequence in string", token, current);
640     Char escape = *current++;
641     switch (escape)
642     {
643     case '"':
644     decoded += '"';
645     break;
646     case '/':
647     decoded += '/';
648     break;
649     case '\\':
650     decoded += '\\';
651     break;
652     case 'b':
653     decoded += '\b';
654     break;
655     case 'f':
656     decoded += '\f';
657     break;
658     case 'n':
659     decoded += '\n';
660     break;
661     case 'r':
662     decoded += '\r';
663     break;
664     case 't':
665     decoded += '\t';
666     break;
667     case 'u':
668     {
669     unsigned int unicode;
670     if (!decodeUnicodeCodePoint (token, current, end, unicode))
671     return false;
672     decoded += codePointToUTF8 (unicode);
673     break;
674     }
675     default:
676     return addError ("Bad escape sequence in string", token, current);
677     }
678     }
679     else
680     decoded += c;
681     }
682     return true;
683     }
684    
685     bool
686     Reader::decodeUnicodeCodePoint (Token &token, Location &current, Location end, unsigned int &unicode)
687     {
688     if (!decodeUnicodeEscapeSequence (token, current, end, unicode))
689     return false;
690     if (unicode >= 0xD800 && unicode <= 0xDBFF)
691     {
692     /* surrogate pairs */
693     if (end - current < 6)
694     return addError ("additional six characters expected to parse unicode surrogate pair.", token, current);
695     unsigned int surrogatePair;
696     if (*(current++) == '\\' && *(current++) == 'u')
697     {
698     if (decodeUnicodeEscapeSequence (token, current, end, surrogatePair))
699     unicode = 0x10000 + ((unicode & 0x3FF) << 10) + (surrogatePair & 0x3FF);
700     else
701     return false;
702     }
703     else
704     return addError ("expecting another \\u token to begin the second half of a unicode surrogate pair", token, current);
705     }
706     return true;
707     }
708    
709     bool
710     Reader::decodeUnicodeEscapeSequence (Token &token, Location &current, Location end, unsigned int &unicode)
711     {
712     if (end - current < 4)
713     return addError ("Bad unicode escape sequence in string: four digits expected.", token, current);
714     unicode = 0;
715     for (int index = 0; index < 4; ++index)
716     {
717     Char c = *current++;
718     unicode *= 16;
719     if (c >= '0' && c <= '9')
720     unicode += c - '0';
721     else if (c >= 'a' && c <= 'f')
722     unicode += c - 'a' + 10;
723     else if (c >= 'A' && c <= 'F')
724     unicode += c - 'A' + 10;
725     else
726     return addError ("Bad unicode escape sequence in string: hexadecimal digit expected.", token, current);
727     }
728     return true;
729     }
730    
731     bool
732     Reader::addError (const std::string &message, Token &token, Location extra)
733     {
734     ErrorInfo info;
735    
736     info.token_ = token;
737     info.message_ = message;
738     info.extra_ = extra;
739     errors_.push_back (info);
740     return false;
741     }
742    
743     bool
744     Reader::recoverFromError (TokenType skipUntilToken)
745     {
746     int errorCount = int(errors_.size ());
747     Token skip;
748    
749     while (true)
750     {
751     if (!readToken (skip))
752     errors_.resize (errorCount); /* discard errors caused by recovery */
753     if (skip.type_ == skipUntilToken || skip.type_ == tokenEndOfStream)
754     break;
755     }
756     errors_.resize (errorCount);
757     return false;
758     }
759    
760     bool
761     Reader::addErrorAndRecover (const std::string &message, Token &token, TokenType skipUntilToken)
762     {
763     addError (message, token);
764     return recoverFromError (skipUntilToken);
765     }
766    
767     Value &
768     Reader::currentValue ()
769     {
770     return *(nodes_.top ());
771     }
772    
773     Reader::Char
774     Reader::getNextChar ()
775     {
776     if (current_ == end_)
777     return 0;
778     return *current_++;
779     }
780    
781     void
782     Reader::getLocationLineAndColumn (Location location, int &line, int &column) const
783     {
784     Location current = begin_;
785     Location lastLineStart = current;
786    
787     line = 0;
788     while (current < location && current != end_)
789     {
790     Char c = *current++;
791     if (c == '\r')
792     {
793     if (*current == '\n')
794     ++current;
795     lastLineStart = current;
796     ++line;
797     }
798     else if (c == '\n')
799     {
800     lastLineStart = current;
801     ++line;
802     }
803     }
804     /* column & line start at 1 */
805     column = int(location - lastLineStart) + 1;
806     ++line;
807     }
808    
809     std::string
810     Reader::getLocationLineAndColumn (Location location) const
811     {
812     int line, column;
813    
814     getLocationLineAndColumn (location, line, column);
815     char buffer[18 + 16 + 16 + 1];
816     sprintf (buffer, "Line %d, Column %d", line, column);
817     return buffer;
818     }
819    
820     std::string
821     Reader::getFormatedErrorMessages () const
822     {
823     std::string formattedMessage;
824    
825     for (Errors::const_iterator itError = errors_.begin ();
826     itError != errors_.end ();
827     ++itError)
828     {
829     const ErrorInfo &error = *itError;
830     formattedMessage += "* " + getLocationLineAndColumn (error.token_.start_) + "\n";
831     formattedMessage += " " + error.message_ + "\n";
832     if (error.extra_)
833     formattedMessage += "See " + getLocationLineAndColumn (error.extra_) + " for detail.\n";
834     }
835     return formattedMessage;
836     }
837    
838     std::istream &
839     operator >> (std::istream &sin, Value &root)
840     {
841     Json::Reader reader;
842     bool ok = reader.parse (sin, root, true);
843    
844 sf-pippijn 1.2 if (!ok)
845     throw std::runtime_error (reader.getFormatedErrorMessages ());
846 sf-pippijn 1.1 return sin;
847     }
848    
849     } // namespace Json