ViewVC Help
View File | Revision Log | Show Annotations | Download File
/cvs/cvsroot/ermyth/modules/rpc/json/reader.C
Revision: 1.1
Committed: Thu Jul 19 08:24:56 2007 UTC (19 years, 2 months ago) by pippijn
Content type: text/plain
Branch: MAIN
Log Message:
initial import. the most important changes since Atheme are:
- fixed many memory leaks
- fixed many bugs
- converted to C++ and use more STL containers
- added a (not very enhanced yet) perl module
- greatly improved XML-RPC speed
- added a JSON-RPC module with code from json-cpp
- added a valgrind memcheck module to operserv
- added a more object oriented base64 implementation
- added a specialised unit test framework
- improved stability
- use gettimeofday() if available
- reworked adding/removing commands
- MemoServ IGNORE DEL can now remove indices

File Contents

# User Rev Content
1 pippijn 1.1 //>>>>>>>>>>> reader.C <<<<<<<<<<<//
2     #include "json/reader.h"
3     #include "json/value.h"
4    
5     #include <utility>
6     #include <iostream>
7     #include <stdexcept>
8    
9     #include <cassert>
10     #include <cstdio>
11    
12     namespace json
13     {
14     static bool
15     in (char c, char c1, char c2, char c3, char c4)
16     {
17     return c == c1 || c == c2 || c == c3 || c == c4;
18     }
19    
20     static bool
21     in (char c, char c1, char c2, char c3, char c4, char c5)
22     {
23     return c == c1 || c == c2 || c == c3 || c == c4 || c == c5;
24     }
25    
26    
27     static bool
28     containsNewLine (Reader::Location begin,
29     Reader::Location end)
30     {
31     for (;begin < end; ++begin)
32     if (*begin == '\012' || *begin == '\015')
33     return true;
34     return false;
35     }
36    
37    
38     // Class Reader
39     Reader::Reader ()
40     {
41     }
42    
43     bool
44     Reader::parse (const std::string &document,
45     Value &root,
46     bool collectComments)
47     {
48     document_ = document;
49     char const *begin = document_.c_str ();
50     char const *end = begin + document_.length ();
51     return parse (begin, end, root, collectComments);
52     }
53    
54     bool
55     Reader::parse (std::istream& sin,
56     Value &root,
57     bool collectComments)
58     {
59     #if 0
60     std::istream_iterator<char> begin (sin);
61     std::istream_iterator<char> end;
62     #endif
63     // Those would allow streamed input from a file, if parse () were a
64     // template function.
65    
66     // Since std::string is reference-counted, this at least does not
67     // create an extra copy.
68     std::string doc;
69     std::getline (sin, doc, (char)EOF);
70     return parse (doc, root, collectComments);
71     }
72    
73     bool
74     Reader::parse (char const *beginDoc, char const *endDOc,
75     Value &root,
76     bool collectComments)
77     {
78     begin_ = beginDoc;
79     end_ = endDOc;
80     collectComments_ = collectComments;
81     current_ = begin_;
82     lastValueEnd_ = 0;
83     lastValue_ = 0;
84     commentsBefore_ = "";
85     errors_.clear ();
86     while (!nodes_.empty ())
87     nodes_.pop ();
88     nodes_.push (&root);
89    
90     bool successful = readValue ();
91     Token token;
92     skipCommentTokens (token);
93     if (collectComments_ && !commentsBefore_.empty ())
94     root.setComment (commentsBefore_, commentAfter);
95     return successful;
96     }
97    
98    
99     bool
100     Reader::readValue ()
101     {
102     Token token;
103     skipCommentTokens (token);
104     bool successful = true;
105    
106     if (collectComments_ && !commentsBefore_.empty ())
107     {
108     currentValue ().setComment (commentsBefore_, commentBefore);
109     commentsBefore_ = "";
110     }
111    
112    
113     switch (token.type_)
114     {
115     case tokenObjectBegin:
116     successful = readObject ();
117     break;
118     case tokenArrayBegin:
119     successful = readArray ();
120     break;
121     case tokenNumber:
122     successful = decodeNumber (token);
123     break;
124     case tokenString:
125     successful = decodeString (token);
126     break;
127     case tokenTrue:
128     currentValue () = true;
129     break;
130     case tokenFalse:
131     currentValue () = false;
132     break;
133     case tokenNull:
134     currentValue () = Value ();
135     break;
136     default:
137     return addError ("Syntax error: value, object or array expected.", token);
138     }
139    
140     if (collectComments_)
141     {
142     lastValueEnd_ = current_;
143     lastValue_ = &currentValue ();
144     }
145    
146     return successful;
147     }
148    
149    
150     void
151     Reader::skipCommentTokens (Token &token)
152     {
153     do
154     {
155     readToken (token);
156     }
157     while (token.type_ == tokenComment);
158     }
159    
160    
161     bool
162     Reader::expectToken (TokenType type, Token &token, char const *message)
163     {
164     readToken (token);
165     if (token.type_ != type)
166     return addError (message, token);
167     return true;
168     }
169    
170    
171     bool
172     Reader::readToken (Token &token)
173     {
174     skipSpaces ();
175     token.start_ = current_;
176     char c = getNextChar ();
177     bool ok = true;
178     switch (c)
179     {
180     case '{':
181     token.type_ = tokenObjectBegin;
182     break;
183     case '}':
184     token.type_ = tokenObjectEnd;
185     break;
186     case '[':
187     token.type_ = tokenArrayBegin;
188     break;
189     case ']':
190     token.type_ = tokenArrayEnd;
191     break;
192     case '"':
193     token.type_ = tokenString;
194     ok = readString ();
195     break;
196     case '/':
197     token.type_ = tokenComment;
198     ok = readComment ();
199     break;
200     #if 0
201     #ifdef __GNUC__
202     case '0'...'9':
203     #endif
204     #else
205     case '0': case '1': case '2': case '3':
206     case '4': case '5': case '6': case '7':
207     case '8': case '9':
208     #endif
209     case '-':
210     token.type_ = tokenNumber;
211     readNumber ();
212     break;
213     case 't':
214     token.type_ = tokenTrue;
215     ok = match ("rue", 3);
216     break;
217     case 'f':
218     token.type_ = tokenFalse;
219     ok = match ("alse", 4);
220     break;
221     case 'n':
222     token.type_ = tokenNull;
223     ok = match ("ull", 3);
224     break;
225     case ',':
226     token.type_ = tokenArraySeparator;
227     break;
228     case ':':
229     token.type_ = tokenMemberSeparator;
230     break;
231     case 0:
232     token.type_ = tokenEndOfStream;
233     break;
234     default:
235     ok = false;
236     break;
237     }
238     if (!ok)
239     token.type_ = tokenError;
240     token.end_ = current_;
241     return true;
242     }
243    
244    
245     void
246     Reader::skipSpaces ()
247     {
248     while (current_ != end_)
249     {
250     char c = *current_;
251     if (c == ' ' || c == '\t' || c == '\r' || c == '\n')
252     ++current_;
253     else
254     break;
255     }
256     }
257    
258    
259     bool
260     Reader::match (Location pattern, int patternLength)
261     {
262     if (end_ - current_ < patternLength)
263     return false;
264     int index = patternLength;
265     while (index--)
266     if (current_[index] != pattern[index])
267     return false;
268     current_ += patternLength;
269     return true;
270     }
271    
272    
273     bool
274     Reader::readComment ()
275     {
276     Location commentBegin = current_ - 1;
277     char c = getNextChar ();
278     bool successful = false;
279     if (c == '*')
280     successful = readCStyleComment ();
281     else if (c == '/')
282     successful = readCppStyleComment ();
283     if (!successful)
284     return false;
285    
286     if (collectComments_)
287     {
288     CommentPlacement placement = commentBefore;
289     if (lastValueEnd_ && !containsNewLine (lastValueEnd_, commentBegin))
290     {
291     if (c != '*' || !containsNewLine (commentBegin, current_))
292     placement = commentAfterOnSameLine;
293     }
294    
295     addComment (commentBegin, current_, placement);
296     }
297     return true;
298     }
299    
300    
301     void
302     Reader::addComment (Location begin,
303     Location end,
304     CommentPlacement placement)
305     {
306     assert (collectComments_);
307     if (placement == commentAfterOnSameLine)
308     {
309     assert (lastValue_ != 0);
310     lastValue_->setComment (std::string (begin, end), placement);
311     }
312     else
313     {
314     if (!commentsBefore_.empty ())
315     commentsBefore_ += "\n";
316     commentsBefore_ += std::string (begin, end);
317     }
318     }
319    
320    
321     bool
322     Reader::readCStyleComment ()
323     {
324     while (current_ != end_)
325     {
326     char c = getNextChar ();
327     if (c == '*' && *current_ == '/')
328     break;
329     }
330     return getNextChar () == '/';
331     }
332    
333    
334     bool
335     Reader::readCppStyleComment ()
336     {
337     while (current_ != end_)
338     {
339     char c = getNextChar ();
340     if (c == '\r' || c == '\n')
341     break;
342     }
343     return true;
344     }
345    
346    
347     void
348     Reader::readNumber ()
349     {
350     while (current_ != end_)
351     {
352     if (!(*current_ >= '0' && *current_ <= '9') &&
353     !in (*current_, '.', 'e', 'E', '+', '-'))
354     break;
355     ++current_;
356     }
357     }
358    
359     bool
360     Reader::readString ()
361     {
362     char c = 0;
363     while (current_ != end_)
364     {
365     c = getNextChar ();
366     if (c == '\\')
367     getNextChar ();
368     else if (c == '"')
369     break;
370     }
371     return c == '"';
372     }
373    
374    
375     bool
376     Reader::readObject ()
377     {
378     Token tokenName;
379     std::string name;
380     currentValue () = Value (objectValue);
381     while (readToken (tokenName))
382     {
383     bool initialTokenOk = true;
384     while (tokenName.type_ == tokenComment && initialTokenOk)
385     initialTokenOk = readToken (tokenName);
386     if (!initialTokenOk)
387     break;
388     if (tokenName.type_ == tokenObjectEnd && name.empty ()) // empty object
389     return true;
390     if (tokenName.type_ != tokenString)
391     break;
392    
393     name = "";
394     if (!decodeString (tokenName, name))
395     return recoverFromError (tokenObjectEnd);
396    
397     Token colon;
398     if (!readToken (colon) || colon.type_ != tokenMemberSeparator)
399     {
400     return addErrorAndRecover ("Missing ':' after object member name",
401     colon,
402     tokenObjectEnd);
403     }
404     Value &value = currentValue ()[ name ];
405     nodes_.push (&value);
406     bool ok = readValue ();
407     nodes_.pop ();
408     if (!ok) // error already set
409     return recoverFromError (tokenObjectEnd);
410    
411     Token comma;
412     if (!readToken (comma)
413     || (comma.type_ != tokenObjectEnd &&
414     comma.type_ != tokenArraySeparator &&
415     comma.type_ != tokenComment))
416     {
417     return addErrorAndRecover ("Missing ',' or '}' in object declaration",
418     comma,
419     tokenObjectEnd);
420     }
421     bool finalizeTokenOk = true;
422     while (comma.type_ == tokenComment &&
423     finalizeTokenOk)
424     finalizeTokenOk = readToken (comma);
425     if (comma.type_ == tokenObjectEnd)
426     return true;
427     }
428     return addErrorAndRecover ("Missing '}' or object member name",
429     tokenName,
430     tokenObjectEnd);
431     }
432    
433    
434     bool
435     Reader::readArray ()
436     {
437     currentValue () = Value (arrayValue);
438     skipSpaces ();
439     if (*current_ == ']') // empty array
440     {
441     Token endArray;
442     readToken (endArray);
443     return true;
444     }
445     int index = 0;
446     while (true)
447     {
448     Value &value = currentValue ()[ index++ ];
449     nodes_.push (&value);
450     bool ok = readValue ();
451     nodes_.pop ();
452     if (!ok) // error already set
453     return recoverFromError (tokenArrayEnd);
454    
455     Token token;
456     if (!readToken (token)
457     || (token.type_ != tokenArraySeparator &&
458     token.type_ != tokenArrayEnd))
459     {
460     return addErrorAndRecover ("Missing ',' or ']' in array declaration",
461     token,
462     tokenArrayEnd);
463     }
464     if (token.type_ == tokenArrayEnd)
465     break;
466     }
467     return true;
468     }
469    
470    
471     bool
472     Reader::decodeNumber (Token &token)
473     {
474     bool isDouble = false;
475     for (Location inspect = token.start_; inspect != token.end_; ++inspect)
476     {
477     isDouble = isDouble
478     || in (*inspect, '.', 'e', 'E', '+')
479     || (*inspect == '-' && inspect != token.start_);
480     }
481     if (isDouble)
482     return decodeDouble (token);
483     Location current = token.start_;
484     bool isNegative = *current == '-';
485     if (isNegative)
486     ++current;
487     unsigned threshold = (isNegative ? unsigned (-Value::minInt)
488     : Value::maxUInt) / 10;
489     unsigned value = 0;
490     while (current < token.end_)
491     {
492     char c = *current++;
493     if (c < '0' || c > '9')
494     return addError ("'" + std::string (token.start_, token.end_) + "' is not a number.", token);
495     if (value >= threshold)
496     return decodeDouble (token);
497     value = value * 10 + unsigned (c - '0');
498     }
499     if (isNegative)
500     currentValue () = -int (value);
501     else if (value <= unsigned (Value::maxInt))
502     currentValue () = int (value);
503     else
504     currentValue () = value;
505     return true;
506     }
507    
508    
509     bool
510     Reader::decodeDouble (Token &token)
511     {
512     double value = 0;
513     const int bufferSize = 32;
514     int count;
515     int length = int (token.end_ - token.start_);
516     if (length <= bufferSize)
517     {
518     char buffer[bufferSize];
519     memcpy (buffer, token.start_, length);
520     buffer[length] = 0;
521     count = sscanf (buffer, "%lf", &value);
522     }
523     else
524     {
525     std::string buffer (token.start_, token.end_);
526     count = sscanf (buffer.c_str (), "%lf", &value);
527     }
528    
529     if (count != 1)
530     return addError ("'" + std::string (token.start_, token.end_) + "' is not a number.", token);
531     currentValue () = value;
532     return true;
533     }
534    
535    
536     bool
537     Reader::decodeString (Token &token)
538     {
539     std::string decoded;
540     if (!decodeString (token, decoded))
541     return false;
542     currentValue () = decoded;
543     return true;
544     }
545    
546    
547     bool
548     Reader::decodeString (Token &token, std::string &decoded)
549     {
550     Location current = token.start_ + 1; // skip '"'
551     Location end = token.end_ - 1; // do not include '"'
552     decoded.reserve (long (end - current));
553    
554     while (current != end)
555     {
556     char c = *current++;
557     if (expect_false (c == '"'))
558     break;
559     else if (expect_false (c == '\\'))
560     {
561     if (expect_false (current == end))
562     return addError ("Empty escape sequence in string", token, current);
563     char escape = *current++;
564     switch (escape)
565     {
566     case '"':
567     case '/':
568     case '\\': decoded += escape; break;
569    
570     case 'b': decoded += '\010'; break;
571     case 't': decoded += '\011'; break;
572     case 'n': decoded += '\012'; break;
573     case 'f': decoded += '\014'; break;
574     case 'r': decoded += '\015'; break;
575     case 'u':
576     {
577     unsigned unicode;
578     if (!decodeUnicodeEscapeSequence (token, current, end, unicode))
579     return false;
580     // @todo encode unicode as utf8.
581     // @todo remember to alter the writer too.
582     }
583     break;
584     default:
585     return addError ("Bad escape sequence in string", token, current);
586     }
587     }
588     else
589     {
590     decoded += c;
591     }
592     }
593    
594     return true;
595     }
596    
597    
598     bool
599     Reader::decodeUnicodeEscapeSequence (Token &token,
600     Location &current,
601     Location end,
602     unsigned &unicode)
603     {
604     if (end - current < 4)
605     return addError ("Bad unicode escape sequence in string: four digits expected.", token, current);
606     unicode = 0;
607     for (int index = 0; index < 4; ++index)
608     {
609     char c = *current++;
610     unicode *= 16;
611     if (c >= '0' && c <= '9')
612     unicode += c - '0';
613     else if (c >= 'a' && c <= 'f')
614     unicode += c - 'a' + 10;
615     else if (c >= 'A' && c <= 'F')
616     unicode += c - 'A' + 10;
617     else
618     return addError ("Bad unicode escape sequence in string: hexadecimal digit expected.", token, current);
619     }
620     return true;
621     }
622    
623    
624     bool
625     Reader::addError (const std::string &message,
626     Token &token,
627     Location extra)
628     {
629     ErrorInfo info;
630     info.token_ = token;
631     info.message_ = message;
632     info.extra_ = extra;
633     errors_.push_back (info);
634     return false;
635     }
636    
637    
638     bool
639     Reader::recoverFromError (TokenType skipUntilToken)
640     {
641     int errorCount = int (errors_.size ());
642     Token skip;
643     while (true)
644     {
645     if (!readToken (skip))
646     errors_.resize (errorCount); // discard errors caused by recovery
647     if (skip.type_ == skipUntilToken || skip.type_ == tokenEndOfStream)
648     break;
649     }
650     errors_.resize (errorCount);
651     return false;
652     }
653    
654    
655     bool
656     Reader::addErrorAndRecover (const std::string &message,
657     Token &token,
658     TokenType skipUntilToken)
659     {
660     addError (message, token);
661     return recoverFromError (skipUntilToken);
662     }
663    
664    
665     Value &
666     Reader::currentValue ()
667     {
668     return *(nodes_.top ());
669     }
670    
671    
672     char
673     Reader::getNextChar ()
674     {
675     if (current_ == end_)
676     return 0;
677     return *current_++;
678     }
679    
680    
681     void
682     Reader::getLocationLineAndColumn (Location location,
683     int &line,
684     int &column) const
685     {
686     Location current = begin_;
687     Location lastLineStart = current;
688     line = 0;
689     while (current < location && current != end_)
690     {
691     char c = *current++;
692     if (c == '\r')
693     {
694     if (*current == '\n')
695     ++current;
696     lastLineStart = current;
697     ++line;
698     }
699     else if (c == '\n')
700     {
701     lastLineStart = current;
702     ++line;
703     }
704     }
705     // column & line start at 1
706     column = int (location - lastLineStart) + 1;
707     ++line;
708     }
709    
710    
711     std::string
712     Reader::getLocationLineAndColumn (Location location) const
713     {
714     int line, column;
715     getLocationLineAndColumn (location, line, column);
716     char buffer[18+16+16+1];
717     sprintf (buffer, "Line %d, Column %d", line, column);
718     return buffer;
719     }
720    
721    
722     std::string
723     Reader::error_msgs () const
724     {
725     std::string formattedMessage;
726     for (Errors::const_iterator itError = errors_.begin ();
727     itError != errors_.end ();
728     ++itError)
729     {
730     const ErrorInfo &error = *itError;
731     formattedMessage += "* " + getLocationLineAndColumn (error.token_.start_) + "\n";
732     formattedMessage += " " + error.message_ + "\n";
733     if (error.extra_)
734     formattedMessage += "See " + getLocationLineAndColumn (error.extra_) + " for detail.\n";
735     }
736     return formattedMessage;
737     }
738    
739    
740     std::istream& operator >> (std::istream &sin, Value &root)
741     {
742     Reader reader;
743     bool ok = reader.parse (sin, root, true);
744     #if 0
745     throw_unless (ok);
746     #endif
747     if (!ok) throw std::runtime_error (reader.error_msgs ());
748     return sin;
749     }
750     } // namespace json