ViewVC Help
View File | Revision Log | Show Annotations | Download File
/cvs/cvsroot/cf.schmorp.de/dclient/src/backend/json/json_reader.cpp
Revision: 1.2
Committed: Mon Oct 18 16:38:02 2010 UTC (15 years, 11 months ago) by sf-pippijn
Branch: MAIN
CVS Tags: HEAD
Changes since 1.1: +2 -2 lines
Log Message:
support msg colours

File Contents

# Content
1 #include <json/reader.h>
2 #include <json/value.h>
3
4 #include <cassert>
5 #include <cstdio>
6 #include <cstring>
7 #include <iostream>
8 #include <stdexcept>
9 #include <utility>
10
11 #if _MSC_VER >= 1400 /* VC++ 8.0 */
12 #pragma warning( disable : 4996 ) /* disable warning about strdup being deprecated. */
13 #endif
14
15 namespace Json
16 {
17
18 /*
19 * Implementation of class Features
20 * ////////////////////////////////
21 */
22
23 Features::Features ()
24 : allowComments_ (true)
25 , strictRoot_ (false)
26 {
27 }
28
29 Features
30 Features::all ()
31 {
32 return Features ();
33 }
34
35 Features
36 Features::strictMode ()
37 {
38 Features features;
39
40 features.allowComments_ = false;
41 features.strictRoot_ = true;
42 return features;
43 }
44
45 /*
46 * Implementation of class Reader
47 * ////////////////////////////////
48 */
49
50
51 static inline bool
52 in (Reader::Char c, Reader::Char c1, Reader::Char c2, Reader::Char c3, Reader::Char c4)
53 {
54 return c == c1 || c == c2 || c == c3 || c == c4;
55 }
56
57 static inline bool
58 in (Reader::Char c, Reader::Char c1, Reader::Char c2, Reader::Char c3, Reader::Char c4, Reader::Char c5)
59 {
60 return c == c1 || c == c2 || c == c3 || c == c4 || c == c5;
61 }
62
63 static bool
64 containsNewLine (Reader::Location begin, Reader::Location end)
65 {
66 for (; begin < end; ++begin)
67 if (*begin == '\n' || *begin == '\r')
68 return true;
69 return false;
70 }
71
72 static std::string
73 codePointToUTF8 (unsigned int cp)
74 {
75 std::string result;
76
77 /* based on description from http://en.wikipedia.org/wiki/UTF-8 */
78
79 if (cp <= 0x7f)
80 {
81 result.resize (1);
82 result[0] = static_cast<char> (cp);
83 }
84 else if (cp <= 0x7FF)
85 {
86 result.resize (2);
87 result[1] = static_cast<char> (0x80 | (0x3f & cp));
88 result[0] = static_cast<char> (0xC0 | (0x1f & (cp >> 6)));
89 }
90 else if (cp <= 0xFFFF)
91 {
92 result.resize (3);
93 result[2] = static_cast<char> (0x80 | (0x3f & cp));
94 result[1] = 0x80 | static_cast<char> ((0x3f & (cp >> 6)));
95 result[0] = 0xE0 | static_cast<char> ((0xf & (cp >> 12)));
96 }
97 else if (cp <= 0x10FFFF)
98 {
99 result.resize (4);
100 result[3] = static_cast<char> (0x80 | (0x3f & cp));
101 result[2] = static_cast<char> (0x80 | (0x3f & (cp >> 6)));
102 result[1] = static_cast<char> (0x80 | (0x3f & (cp >> 12)));
103 result[0] = static_cast<char> (0xF0 | (0x7 & (cp >> 18)));
104 }
105
106 return result;
107 }
108
109 /*
110 * Class Reader
111 * //////////////////////////////////////////////////////////////////
112 */
113
114 Reader::Reader ()
115 : features_ (Features::all ())
116 {
117 }
118
119 Reader::Reader (const Features &features)
120 : features_ (features)
121 {
122 }
123
124 bool
125 Reader::parse (const std::string &document, Value &root, bool collectComments)
126 {
127 document_ = document;
128 const char *begin = document_.c_str ();
129 const char *end = begin + document_.length ();
130 return parse (begin, end, root, collectComments);
131 }
132
133 bool
134 Reader::parse (std::istream &sin, Value &root, bool collectComments)
135 {
136 /*
137 * std::istream_iterator<char> begin(sin);
138 * std::istream_iterator<char> end;
139 * Those would allow streamed input from a file, if parse() were a
140 * template function.
141 */
142
143 /*
144 * Since std::string is reference-counted, this at least does not
145 * create an extra copy.
146 */
147 std::string doc;
148
149 std::getline (sin, doc, (char)EOF);
150 return parse (doc, root, collectComments);
151 }
152
153 bool
154 Reader::parse (const char *beginDoc, const char *endDoc, Value &root, bool collectComments)
155 {
156 if (!features_.allowComments_)
157 collectComments = false;
158
159 begin_ = beginDoc;
160 end_ = endDoc;
161 collectComments_ = collectComments;
162 current_ = begin_;
163 lastValueEnd_ = 0;
164 lastValue_ = 0;
165 commentsBefore_ = "";
166 errors_.clear ();
167 while (!nodes_.empty ())
168 nodes_.pop ();
169 nodes_.push (&root);
170
171 bool successful = readValue ();
172 Token token;
173 skipCommentTokens (token);
174 if (collectComments_ && !commentsBefore_.empty ())
175 root.setComment (commentsBefore_, commentAfter);
176 if (features_.strictRoot_)
177 if (!root.isArray () && !root.isObject ())
178 {
179 /* Set error location to start of doc, ideally should be first token found in doc */
180 token.type_ = tokenError;
181 token.start_ = beginDoc;
182 token.end_ = endDoc;
183 addError ("A valid JSON document must be either an array or an object value.",
184 token);
185 return false;
186 }
187 return successful;
188 }
189
190 bool
191 Reader::readValue ()
192 {
193 Token token;
194
195 skipCommentTokens (token);
196 bool successful = true;
197
198 if (collectComments_ && !commentsBefore_.empty ())
199 {
200 currentValue ().setComment (commentsBefore_, commentBefore);
201 commentsBefore_ = "";
202 }
203
204
205 switch (token.type_)
206 {
207 case tokenObjectBegin:
208 successful = readObject (token);
209 break;
210 case tokenArrayBegin:
211 successful = readArray (token);
212 break;
213 case tokenNumber:
214 successful = decodeNumber (token);
215 break;
216 case tokenString:
217 successful = decodeString (token);
218 break;
219 case tokenTrue:
220 currentValue () = true;
221 break;
222 case tokenFalse:
223 currentValue () = false;
224 break;
225 case tokenNull:
226 currentValue () = Value ();
227 break;
228 default:
229 return addError ("Syntax error: value, object or array expected.", token);
230 }
231
232 if (collectComments_)
233 {
234 lastValueEnd_ = current_;
235 lastValue_ = &currentValue ();
236 }
237
238 return successful;
239 }
240
241 void
242 Reader::skipCommentTokens (Token &token)
243 {
244 if (features_.allowComments_)
245 {
246 do
247 readToken (token);
248 while (token.type_ == tokenComment);
249 }
250 else
251 readToken (token);
252 }
253
254 bool
255 Reader::expectToken (TokenType type, Token &token, const char *message)
256 {
257 readToken (token);
258 if (token.type_ != type)
259 return addError (message, token);
260 return true;
261 }
262
263 bool
264 Reader::readToken (Token &token)
265 {
266 skipSpaces ();
267 token.start_ = current_;
268 Char c = getNextChar ();
269 bool ok = true;
270 switch (c)
271 {
272 case '{':
273 token.type_ = tokenObjectBegin;
274 break;
275 case '}':
276 token.type_ = tokenObjectEnd;
277 break;
278 case '[':
279 token.type_ = tokenArrayBegin;
280 break;
281 case ']':
282 token.type_ = tokenArrayEnd;
283 break;
284 case '"':
285 token.type_ = tokenString;
286 ok = readString ();
287 break;
288 case '/':
289 token.type_ = tokenComment;
290 ok = readComment ();
291 break;
292 case '0':
293 case '1':
294 case '2':
295 case '3':
296 case '4':
297 case '5':
298 case '6':
299 case '7':
300 case '8':
301 case '9':
302 case '-':
303 token.type_ = tokenNumber;
304 readNumber ();
305 break;
306 case 't':
307 token.type_ = tokenTrue;
308 ok = match ("rue", 3);
309 break;
310 case 'f':
311 token.type_ = tokenFalse;
312 ok = match ("alse", 4);
313 break;
314 case 'n':
315 token.type_ = tokenNull;
316 ok = match ("ull", 3);
317 break;
318 case ',':
319 token.type_ = tokenArraySeparator;
320 break;
321 case ':':
322 token.type_ = tokenMemberSeparator;
323 break;
324 case 0:
325 token.type_ = tokenEndOfStream;
326 break;
327 default:
328 ok = false;
329 break;
330 }
331 if (!ok)
332 token.type_ = tokenError;
333 token.end_ = current_;
334 return true;
335 }
336
337 void
338 Reader::skipSpaces ()
339 {
340 while (current_ != end_)
341 {
342 Char c = *current_;
343 if (c == ' ' || c == '\t' || c == '\r' || c == '\n')
344 ++current_;
345 else
346 break;
347 }
348 }
349
350 bool
351 Reader::match (Location pattern, int patternLength)
352 {
353 if (end_ - current_ < patternLength)
354 return false;
355 int index = patternLength;
356 while (index--)
357 if (current_[index] != pattern[index])
358 return false;
359 current_ += patternLength;
360 return true;
361 }
362
363 bool
364 Reader::readComment ()
365 {
366 Location commentBegin = current_ - 1;
367 Char c = getNextChar ();
368 bool successful = false;
369
370 if (c == '*')
371 successful = readCStyleComment ();
372 else if (c == '/')
373 successful = readCppStyleComment ();
374 if (!successful)
375 return false;
376
377 if (collectComments_)
378 {
379 CommentPlacement placement = commentBefore;
380 if (lastValueEnd_ && !containsNewLine (lastValueEnd_, commentBegin))
381 if (c != '*' || !containsNewLine (commentBegin, current_))
382 placement = commentAfterOnSameLine;
383
384 addComment (commentBegin, current_, placement);
385 }
386 return true;
387 }
388
389 void
390 Reader::addComment (Location begin, Location end, CommentPlacement placement)
391 {
392 assert (collectComments_);
393 if (placement == commentAfterOnSameLine)
394 {
395 assert (lastValue_ != 0);
396 lastValue_->setComment (std::string (begin, end), placement);
397 }
398 else
399 {
400 if (!commentsBefore_.empty ())
401 commentsBefore_ += "\n";
402 commentsBefore_ += std::string (begin, end);
403 }
404 }
405
406 bool
407 Reader::readCStyleComment ()
408 {
409 while (current_ != end_)
410 {
411 Char c = getNextChar ();
412 if (c == '*' && *current_ == '/')
413 break;
414 }
415 return getNextChar () == '/';
416 }
417
418 bool
419 Reader::readCppStyleComment ()
420 {
421 while (current_ != end_)
422 {
423 Char c = getNextChar ();
424 if (c == '\r' || c == '\n')
425 break;
426 }
427 return true;
428 }
429
430 void
431 Reader::readNumber ()
432 {
433 while (current_ != end_)
434 {
435 if (!(*current_ >= '0' && *current_ <= '9') &&
436 !in (*current_, '.', 'e', 'E', '+', '-'))
437 break;
438 ++current_;
439 }
440 }
441
442 bool
443 Reader::readString ()
444 {
445 Char c = 0;
446
447 while (current_ != end_)
448 {
449 c = getNextChar ();
450 if (c == '\\')
451 getNextChar ();
452 else if (c == '"')
453 break;
454 }
455 return c == '"';
456 }
457
458 bool
459 Reader::readObject (Token &tokenStart)
460 {
461 Token tokenName;
462 std::string name;
463
464 currentValue () = Value (objectValue);
465 while (readToken (tokenName))
466 {
467 bool initialTokenOk = true;
468 while (tokenName.type_ == tokenComment && initialTokenOk)
469 initialTokenOk = readToken (tokenName);
470 if (!initialTokenOk)
471 break;
472 if (tokenName.type_ == tokenObjectEnd && name.empty ()) /* empty object */
473 return true;
474 if (tokenName.type_ != tokenString)
475 break;
476
477 name = "";
478 if (!decodeString (tokenName, name))
479 return recoverFromError (tokenObjectEnd);
480
481 Token colon;
482 if (!readToken (colon) || colon.type_ != tokenMemberSeparator)
483 return addErrorAndRecover ("Missing ':' after object member name",
484 colon,
485 tokenObjectEnd);
486 Value &value = currentValue ()[name];
487 nodes_.push (&value);
488 bool ok = readValue ();
489 nodes_.pop ();
490 if (!ok) /* error already set */
491 return recoverFromError (tokenObjectEnd);
492
493 Token comma;
494 if (!readToken (comma)
495 || (comma.type_ != tokenObjectEnd &&
496 comma.type_ != tokenArraySeparator &&
497 comma.type_ != tokenComment))
498 return addErrorAndRecover ("Missing ',' or '}' in object declaration",
499 comma,
500 tokenObjectEnd);
501 bool finalizeTokenOk = true;
502 while (comma.type_ == tokenComment &&
503 finalizeTokenOk)
504 finalizeTokenOk = readToken (comma);
505 if (comma.type_ == tokenObjectEnd)
506 return true;
507 }
508 return addErrorAndRecover ("Missing '}' or object member name",
509 tokenName,
510 tokenObjectEnd);
511 }
512
513 bool
514 Reader::readArray (Token &tokenStart)
515 {
516 currentValue () = Value (arrayValue);
517 skipSpaces ();
518 if (*current_ == ']') /* empty array */
519 {
520 Token endArray;
521 readToken (endArray);
522 return true;
523 }
524 int index = 0;
525 while (true)
526 {
527 Value &value = currentValue ()[index++];
528 nodes_.push (&value);
529 bool ok = readValue ();
530 nodes_.pop ();
531 if (!ok) /* error already set */
532 return recoverFromError (tokenArrayEnd);
533
534 Token token;
535 /* Accept Comment after last item in the array. */
536 ok = readToken (token);
537 while (token.type_ == tokenComment && ok)
538 ok = readToken (token);
539 bool badTokenType = (token.type_ == tokenArraySeparator &&
540 token.type_ == tokenArrayEnd);
541 if (!ok || badTokenType)
542 return addErrorAndRecover ("Missing ',' or ']' in array declaration",
543 token,
544 tokenArrayEnd);
545 if (token.type_ == tokenArrayEnd)
546 break;
547 }
548 return true;
549 }
550
551 bool
552 Reader::decodeNumber (Token &token)
553 {
554 bool isDouble = false;
555
556 for (Location inspect = token.start_; inspect != token.end_; ++inspect)
557 isDouble = isDouble
558 || in (*inspect, '.', 'e', 'E', '+')
559 || (*inspect == '-' && inspect != token.start_);
560 if (isDouble)
561 return decodeDouble (token);
562 Location current = token.start_;
563 bool isNegative = *current == '-';
564 if (isNegative)
565 ++current;
566 Value::UInt threshold = (isNegative ? Value::UInt (-Value::minInt)
567 : Value::maxUInt) / 10;
568 Value::UInt value = 0;
569 while (current < token.end_)
570 {
571 Char c = *current++;
572 if (c < '0' || c > '9')
573 return addError ("'" + std::string (token.start_, token.end_) + "' is not a number.", token);
574 if (value >= threshold)
575 return decodeDouble (token);
576 value = value * 10 + Value::UInt (c - '0');
577 }
578 if (isNegative)
579 currentValue () = -Value::Int (value);
580 else if (value <= Value::UInt (Value::maxInt))
581 currentValue () = Value::Int (value);
582 else
583 currentValue () = value;
584 return true;
585 }
586
587 bool
588 Reader::decodeDouble (Token &token)
589 {
590 double value = 0;
591 const int bufferSize = 32;
592 int count;
593 int length = int(token.end_ - token.start_);
594
595 if (length <= bufferSize)
596 {
597 Char buffer[bufferSize];
598 memcpy (buffer, token.start_, length);
599 buffer[length] = 0;
600 count = sscanf (buffer, "%lf", &value);
601 }
602 else
603 {
604 std::string buffer (token.start_, token.end_);
605 count = sscanf (buffer.c_str (), "%lf", &value);
606 }
607
608 if (count != 1)
609 return addError ("'" + std::string (token.start_, token.end_) + "' is not a number.", token);
610 currentValue () = value;
611 return true;
612 }
613
614 bool
615 Reader::decodeString (Token &token)
616 {
617 std::string decoded;
618
619 if (!decodeString (token, decoded))
620 return false;
621 currentValue () = decoded;
622 return true;
623 }
624
625 bool
626 Reader::decodeString (Token &token, std::string &decoded)
627 {
628 decoded.reserve (token.end_ - token.start_ - 2);
629 Location current = token.start_ + 1; /* skip '"' */
630 Location end = token.end_ - 1; /* do not include '"' */
631 while (current != end)
632 {
633 Char c = *current++;
634 if (c == '"')
635 break;
636 else if (c == '\\')
637 {
638 if (current == end)
639 return addError ("Empty escape sequence in string", token, current);
640 Char escape = *current++;
641 switch (escape)
642 {
643 case '"':
644 decoded += '"';
645 break;
646 case '/':
647 decoded += '/';
648 break;
649 case '\\':
650 decoded += '\\';
651 break;
652 case 'b':
653 decoded += '\b';
654 break;
655 case 'f':
656 decoded += '\f';
657 break;
658 case 'n':
659 decoded += '\n';
660 break;
661 case 'r':
662 decoded += '\r';
663 break;
664 case 't':
665 decoded += '\t';
666 break;
667 case 'u':
668 {
669 unsigned int unicode;
670 if (!decodeUnicodeCodePoint (token, current, end, unicode))
671 return false;
672 decoded += codePointToUTF8 (unicode);
673 break;
674 }
675 default:
676 return addError ("Bad escape sequence in string", token, current);
677 }
678 }
679 else
680 decoded += c;
681 }
682 return true;
683 }
684
685 bool
686 Reader::decodeUnicodeCodePoint (Token &token, Location &current, Location end, unsigned int &unicode)
687 {
688 if (!decodeUnicodeEscapeSequence (token, current, end, unicode))
689 return false;
690 if (unicode >= 0xD800 && unicode <= 0xDBFF)
691 {
692 /* surrogate pairs */
693 if (end - current < 6)
694 return addError ("additional six characters expected to parse unicode surrogate pair.", token, current);
695 unsigned int surrogatePair;
696 if (*(current++) == '\\' && *(current++) == 'u')
697 {
698 if (decodeUnicodeEscapeSequence (token, current, end, surrogatePair))
699 unicode = 0x10000 + ((unicode & 0x3FF) << 10) + (surrogatePair & 0x3FF);
700 else
701 return false;
702 }
703 else
704 return addError ("expecting another \\u token to begin the second half of a unicode surrogate pair", token, current);
705 }
706 return true;
707 }
708
709 bool
710 Reader::decodeUnicodeEscapeSequence (Token &token, Location &current, Location end, unsigned int &unicode)
711 {
712 if (end - current < 4)
713 return addError ("Bad unicode escape sequence in string: four digits expected.", token, current);
714 unicode = 0;
715 for (int index = 0; index < 4; ++index)
716 {
717 Char c = *current++;
718 unicode *= 16;
719 if (c >= '0' && c <= '9')
720 unicode += c - '0';
721 else if (c >= 'a' && c <= 'f')
722 unicode += c - 'a' + 10;
723 else if (c >= 'A' && c <= 'F')
724 unicode += c - 'A' + 10;
725 else
726 return addError ("Bad unicode escape sequence in string: hexadecimal digit expected.", token, current);
727 }
728 return true;
729 }
730
731 bool
732 Reader::addError (const std::string &message, Token &token, Location extra)
733 {
734 ErrorInfo info;
735
736 info.token_ = token;
737 info.message_ = message;
738 info.extra_ = extra;
739 errors_.push_back (info);
740 return false;
741 }
742
743 bool
744 Reader::recoverFromError (TokenType skipUntilToken)
745 {
746 int errorCount = int(errors_.size ());
747 Token skip;
748
749 while (true)
750 {
751 if (!readToken (skip))
752 errors_.resize (errorCount); /* discard errors caused by recovery */
753 if (skip.type_ == skipUntilToken || skip.type_ == tokenEndOfStream)
754 break;
755 }
756 errors_.resize (errorCount);
757 return false;
758 }
759
760 bool
761 Reader::addErrorAndRecover (const std::string &message, Token &token, TokenType skipUntilToken)
762 {
763 addError (message, token);
764 return recoverFromError (skipUntilToken);
765 }
766
767 Value &
768 Reader::currentValue ()
769 {
770 return *(nodes_.top ());
771 }
772
773 Reader::Char
774 Reader::getNextChar ()
775 {
776 if (current_ == end_)
777 return 0;
778 return *current_++;
779 }
780
781 void
782 Reader::getLocationLineAndColumn (Location location, int &line, int &column) const
783 {
784 Location current = begin_;
785 Location lastLineStart = current;
786
787 line = 0;
788 while (current < location && current != end_)
789 {
790 Char c = *current++;
791 if (c == '\r')
792 {
793 if (*current == '\n')
794 ++current;
795 lastLineStart = current;
796 ++line;
797 }
798 else if (c == '\n')
799 {
800 lastLineStart = current;
801 ++line;
802 }
803 }
804 /* column & line start at 1 */
805 column = int(location - lastLineStart) + 1;
806 ++line;
807 }
808
809 std::string
810 Reader::getLocationLineAndColumn (Location location) const
811 {
812 int line, column;
813
814 getLocationLineAndColumn (location, line, column);
815 char buffer[18 + 16 + 16 + 1];
816 sprintf (buffer, "Line %d, Column %d", line, column);
817 return buffer;
818 }
819
820 std::string
821 Reader::getFormatedErrorMessages () const
822 {
823 std::string formattedMessage;
824
825 for (Errors::const_iterator itError = errors_.begin ();
826 itError != errors_.end ();
827 ++itError)
828 {
829 const ErrorInfo &error = *itError;
830 formattedMessage += "* " + getLocationLineAndColumn (error.token_.start_) + "\n";
831 formattedMessage += " " + error.message_ + "\n";
832 if (error.extra_)
833 formattedMessage += "See " + getLocationLineAndColumn (error.extra_) + " for detail.\n";
834 }
835 return formattedMessage;
836 }
837
838 std::istream &
839 operator >> (std::istream &sin, Value &root)
840 {
841 Json::Reader reader;
842 bool ok = reader.parse (sin, root, true);
843
844 if (!ok)
845 throw std::runtime_error (reader.getFormatedErrorMessages ());
846 return sin;
847 }
848
849 } // namespace Json