parser.hpp 18 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486
  1. #pragma once
  2. #include <cassert> // assert
  3. #include <cmath> // isfinite
  4. #include <cstdint> // uint8_t
  5. #include <functional> // function
  6. #include <string> // string
  7. #include <utility> // move
  8. #include <nlohmann/detail/exceptions.hpp>
  9. #include <nlohmann/detail/macro_scope.hpp>
  10. #include <nlohmann/detail/input/input_adapters.hpp>
  11. #include <nlohmann/detail/input/json_sax.hpp>
  12. #include <nlohmann/detail/input/lexer.hpp>
  13. #include <nlohmann/detail/value_t.hpp>
  14. namespace nlohmann
  15. {
  16. namespace detail
  17. {
  18. ////////////
  19. // parser //
  20. ////////////
  21. /*!
  22. @brief syntax analysis
  23. This class implements a recursive decent parser.
  24. */
  25. template<typename BasicJsonType>
  26. class parser
  27. {
  28. using number_integer_t = typename BasicJsonType::number_integer_t;
  29. using number_unsigned_t = typename BasicJsonType::number_unsigned_t;
  30. using number_float_t = typename BasicJsonType::number_float_t;
  31. using string_t = typename BasicJsonType::string_t;
  32. using lexer_t = lexer<BasicJsonType>;
  33. using token_type = typename lexer_t::token_type;
  34. public:
  35. enum class parse_event_t : uint8_t
  36. {
  37. /// the parser read `{` and started to process a JSON object
  38. object_start,
  39. /// the parser read `}` and finished processing a JSON object
  40. object_end,
  41. /// the parser read `[` and started to process a JSON array
  42. array_start,
  43. /// the parser read `]` and finished processing a JSON array
  44. array_end,
  45. /// the parser read a key of a value in an object
  46. key,
  47. /// the parser finished reading a JSON value
  48. value
  49. };
  50. using json_sax_t = json_sax<BasicJsonType>;
  51. using parser_callback_t =
  52. std::function<bool(int depth, parse_event_t event, BasicJsonType& parsed)>;
  53. /// a parser reading from an input adapter
  54. explicit parser(detail::input_adapter_t&& adapter,
  55. const parser_callback_t cb = nullptr,
  56. const bool allow_exceptions_ = true)
  57. : callback(cb), m_lexer(std::move(adapter)), allow_exceptions(allow_exceptions_)
  58. {
  59. // read first token
  60. get_token();
  61. }
  62. /*!
  63. @brief public parser interface
  64. @param[in] strict whether to expect the last token to be EOF
  65. @param[in,out] result parsed JSON value
  66. @throw parse_error.101 in case of an unexpected token
  67. @throw parse_error.102 if to_unicode fails or surrogate error
  68. @throw parse_error.103 if to_unicode fails
  69. */
  70. void parse(const bool strict, BasicJsonType& result)
  71. {
  72. if (callback)
  73. {
  74. json_sax_dom_callback_parser<BasicJsonType> sdp(result, callback, allow_exceptions);
  75. sax_parse_internal(&sdp);
  76. result.assert_invariant();
  77. // in strict mode, input must be completely read
  78. if (strict and (get_token() != token_type::end_of_input))
  79. {
  80. sdp.parse_error(m_lexer.get_position(),
  81. m_lexer.get_token_string(),
  82. parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input)));
  83. }
  84. // in case of an error, return discarded value
  85. if (sdp.is_errored())
  86. {
  87. result = value_t::discarded;
  88. return;
  89. }
  90. // set top-level value to null if it was discarded by the callback
  91. // function
  92. if (result.is_discarded())
  93. {
  94. result = nullptr;
  95. }
  96. }
  97. else
  98. {
  99. json_sax_dom_parser<BasicJsonType> sdp(result, allow_exceptions);
  100. sax_parse_internal(&sdp);
  101. result.assert_invariant();
  102. // in strict mode, input must be completely read
  103. if (strict and (get_token() != token_type::end_of_input))
  104. {
  105. sdp.parse_error(m_lexer.get_position(),
  106. m_lexer.get_token_string(),
  107. parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input)));
  108. }
  109. // in case of an error, return discarded value
  110. if (sdp.is_errored())
  111. {
  112. result = value_t::discarded;
  113. return;
  114. }
  115. }
  116. }
  117. /*!
  118. @brief public accept interface
  119. @param[in] strict whether to expect the last token to be EOF
  120. @return whether the input is a proper JSON text
  121. */
  122. bool accept(const bool strict = true)
  123. {
  124. json_sax_acceptor<BasicJsonType> sax_acceptor;
  125. return sax_parse(&sax_acceptor, strict);
  126. }
  127. bool sax_parse(json_sax_t* sax, const bool strict = true)
  128. {
  129. const bool result = sax_parse_internal(sax);
  130. // strict mode: next byte must be EOF
  131. if (result and strict and (get_token() != token_type::end_of_input))
  132. {
  133. return sax->parse_error(m_lexer.get_position(),
  134. m_lexer.get_token_string(),
  135. parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input)));
  136. }
  137. return result;
  138. }
  139. private:
  140. bool sax_parse_internal(json_sax_t* sax)
  141. {
  142. // stack to remember the hieararchy of structured values we are parsing
  143. // true = array; false = object
  144. std::vector<bool> states;
  145. // value to avoid a goto (see comment where set to true)
  146. bool skip_to_state_evaluation = false;
  147. while (true)
  148. {
  149. if (not skip_to_state_evaluation)
  150. {
  151. // invariant: get_token() was called before each iteration
  152. switch (last_token)
  153. {
  154. case token_type::begin_object:
  155. {
  156. if (JSON_UNLIKELY(not sax->start_object()))
  157. {
  158. return false;
  159. }
  160. // closing } -> we are done
  161. if (get_token() == token_type::end_object)
  162. {
  163. if (JSON_UNLIKELY(not sax->end_object()))
  164. {
  165. return false;
  166. }
  167. break;
  168. }
  169. // parse key
  170. if (JSON_UNLIKELY(last_token != token_type::value_string))
  171. {
  172. return sax->parse_error(m_lexer.get_position(),
  173. m_lexer.get_token_string(),
  174. parse_error::create(101, m_lexer.get_position(), exception_message(token_type::value_string)));
  175. }
  176. else
  177. {
  178. if (JSON_UNLIKELY(not sax->key(m_lexer.get_string())))
  179. {
  180. return false;
  181. }
  182. }
  183. // parse separator (:)
  184. if (JSON_UNLIKELY(get_token() != token_type::name_separator))
  185. {
  186. return sax->parse_error(m_lexer.get_position(),
  187. m_lexer.get_token_string(),
  188. parse_error::create(101, m_lexer.get_position(), exception_message(token_type::name_separator)));
  189. }
  190. // remember we are now inside an object
  191. states.push_back(false);
  192. // parse values
  193. get_token();
  194. continue;
  195. }
  196. case token_type::begin_array:
  197. {
  198. if (JSON_UNLIKELY(not sax->start_array()))
  199. {
  200. return false;
  201. }
  202. // closing ] -> we are done
  203. if (get_token() == token_type::end_array)
  204. {
  205. if (JSON_UNLIKELY(not sax->end_array()))
  206. {
  207. return false;
  208. }
  209. break;
  210. }
  211. // remember we are now inside an array
  212. states.push_back(true);
  213. // parse values (no need to call get_token)
  214. continue;
  215. }
  216. case token_type::value_float:
  217. {
  218. const auto res = m_lexer.get_number_float();
  219. if (JSON_UNLIKELY(not std::isfinite(res)))
  220. {
  221. return sax->parse_error(m_lexer.get_position(),
  222. m_lexer.get_token_string(),
  223. out_of_range::create(406, "number overflow parsing '" + m_lexer.get_token_string() + "'"));
  224. }
  225. else
  226. {
  227. if (JSON_UNLIKELY(not sax->number_float(res, m_lexer.get_string())))
  228. {
  229. return false;
  230. }
  231. break;
  232. }
  233. }
  234. case token_type::literal_false:
  235. {
  236. if (JSON_UNLIKELY(not sax->boolean(false)))
  237. {
  238. return false;
  239. }
  240. break;
  241. }
  242. case token_type::literal_null:
  243. {
  244. if (JSON_UNLIKELY(not sax->null()))
  245. {
  246. return false;
  247. }
  248. break;
  249. }
  250. case token_type::literal_true:
  251. {
  252. if (JSON_UNLIKELY(not sax->boolean(true)))
  253. {
  254. return false;
  255. }
  256. break;
  257. }
  258. case token_type::value_integer:
  259. {
  260. if (JSON_UNLIKELY(not sax->number_integer(m_lexer.get_number_integer())))
  261. {
  262. return false;
  263. }
  264. break;
  265. }
  266. case token_type::value_string:
  267. {
  268. if (JSON_UNLIKELY(not sax->string(m_lexer.get_string())))
  269. {
  270. return false;
  271. }
  272. break;
  273. }
  274. case token_type::value_unsigned:
  275. {
  276. if (JSON_UNLIKELY(not sax->number_unsigned(m_lexer.get_number_unsigned())))
  277. {
  278. return false;
  279. }
  280. break;
  281. }
  282. case token_type::parse_error:
  283. {
  284. // using "uninitialized" to avoid "expected" message
  285. return sax->parse_error(m_lexer.get_position(),
  286. m_lexer.get_token_string(),
  287. parse_error::create(101, m_lexer.get_position(), exception_message(token_type::uninitialized)));
  288. }
  289. default: // the last token was unexpected
  290. {
  291. return sax->parse_error(m_lexer.get_position(),
  292. m_lexer.get_token_string(),
  293. parse_error::create(101, m_lexer.get_position(), exception_message(token_type::literal_or_value)));
  294. }
  295. }
  296. }
  297. else
  298. {
  299. skip_to_state_evaluation = false;
  300. }
  301. // we reached this line after we successfully parsed a value
  302. if (states.empty())
  303. {
  304. // empty stack: we reached the end of the hieararchy: done
  305. return true;
  306. }
  307. else
  308. {
  309. if (states.back()) // array
  310. {
  311. // comma -> next value
  312. if (get_token() == token_type::value_separator)
  313. {
  314. // parse a new value
  315. get_token();
  316. continue;
  317. }
  318. // closing ]
  319. if (JSON_LIKELY(last_token == token_type::end_array))
  320. {
  321. if (JSON_UNLIKELY(not sax->end_array()))
  322. {
  323. return false;
  324. }
  325. // We are done with this array. Before we can parse a
  326. // new value, we need to evaluate the new state first.
  327. // By setting skip_to_state_evaluation to false, we
  328. // are effectively jumping to the beginning of this if.
  329. assert(not states.empty());
  330. states.pop_back();
  331. skip_to_state_evaluation = true;
  332. continue;
  333. }
  334. else
  335. {
  336. return sax->parse_error(m_lexer.get_position(),
  337. m_lexer.get_token_string(),
  338. parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_array)));
  339. }
  340. }
  341. else // object
  342. {
  343. // comma -> next value
  344. if (get_token() == token_type::value_separator)
  345. {
  346. // parse key
  347. if (JSON_UNLIKELY(get_token() != token_type::value_string))
  348. {
  349. return sax->parse_error(m_lexer.get_position(),
  350. m_lexer.get_token_string(),
  351. parse_error::create(101, m_lexer.get_position(), exception_message(token_type::value_string)));
  352. }
  353. else
  354. {
  355. if (JSON_UNLIKELY(not sax->key(m_lexer.get_string())))
  356. {
  357. return false;
  358. }
  359. }
  360. // parse separator (:)
  361. if (JSON_UNLIKELY(get_token() != token_type::name_separator))
  362. {
  363. return sax->parse_error(m_lexer.get_position(),
  364. m_lexer.get_token_string(),
  365. parse_error::create(101, m_lexer.get_position(), exception_message(token_type::name_separator)));
  366. }
  367. // parse values
  368. get_token();
  369. continue;
  370. }
  371. // closing }
  372. if (JSON_LIKELY(last_token == token_type::end_object))
  373. {
  374. if (JSON_UNLIKELY(not sax->end_object()))
  375. {
  376. return false;
  377. }
  378. // We are done with this object. Before we can parse a
  379. // new value, we need to evaluate the new state first.
  380. // By setting skip_to_state_evaluation to false, we
  381. // are effectively jumping to the beginning of this if.
  382. assert(not states.empty());
  383. states.pop_back();
  384. skip_to_state_evaluation = true;
  385. continue;
  386. }
  387. else
  388. {
  389. return sax->parse_error(m_lexer.get_position(),
  390. m_lexer.get_token_string(),
  391. parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_object)));
  392. }
  393. }
  394. }
  395. }
  396. }
  397. /// get next token from lexer
  398. token_type get_token()
  399. {
  400. return (last_token = m_lexer.scan());
  401. }
  402. std::string exception_message(const token_type expected)
  403. {
  404. std::string error_msg = "syntax error - ";
  405. if (last_token == token_type::parse_error)
  406. {
  407. error_msg += std::string(m_lexer.get_error_message()) + "; last read: '" +
  408. m_lexer.get_token_string() + "'";
  409. }
  410. else
  411. {
  412. error_msg += "unexpected " + std::string(lexer_t::token_type_name(last_token));
  413. }
  414. if (expected != token_type::uninitialized)
  415. {
  416. error_msg += "; expected " + std::string(lexer_t::token_type_name(expected));
  417. }
  418. return error_msg;
  419. }
  420. private:
  421. /// callback function
  422. const parser_callback_t callback = nullptr;
  423. /// the type of the last read token
  424. token_type last_token = token_type::uninitialized;
  425. /// the lexer
  426. lexer_t m_lexer;
  427. /// whether to throw exceptions in case of errors
  428. const bool allow_exceptions = true;
  429. };
  430. }
  431. }