mirror of
https://github.com/simdjson/simdjson
synced 2026-06-08 17:27:07 +00:00
Compare commits
1541 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 17b03de3cd | |||
| 15297c12d1 | |||
| 2a84f35e3f | |||
| 7939ad5307 | |||
| c67add57d8 | |||
| cbacec0760 | |||
| 82fcf141f7 | |||
| aaa12c8fb6 | |||
| 9b5150daef | |||
| 9304d88920 | |||
| 725ca010e7 | |||
| 59c857e969 | |||
| 45c9136443 | |||
| bc4087ac96 | |||
| d9c4191e8a | |||
| 3fa40b8dc2 | |||
| dc69bc28ae | |||
| 53577f11e1 | |||
| 1abb64f6f6 | |||
| b442b29ccf | |||
| 5609851747 | |||
| 68a8004518 | |||
| 0b39e3a6cf | |||
| 924ad2d592 | |||
| 5202d07a77 | |||
| b632107a7a | |||
| f62ca21dd1 | |||
| 553befa012 | |||
| def624a50c | |||
| 268f26a84f | |||
| d9456b3030 | |||
| f195f849d0 | |||
| 669e5b91b4 | |||
| a06e0958ca | |||
| a56149e79a | |||
| c805fc28a4 | |||
| 218c274090 | |||
| 921c79f26f | |||
| af4db55e66 | |||
| 9f78559cc8 | |||
| 23b4bc93aa | |||
| 6321df078a | |||
| e6258377d9 | |||
| b3a26fd0cc | |||
| 47669566da | |||
| 03a1bb0a5b | |||
| 54ffbbe7db | |||
| 0b82f07115 | |||
| 265db2e533 | |||
| 2b2abf4895 | |||
| f93fb21c95 | |||
| b7fe764e6c | |||
| 500e4c3572 | |||
| 55281c01fb | |||
| a8bf10ea5a | |||
| 9f60093b41 | |||
| 562d4f4f58 | |||
| ac87437588 | |||
| b1444b4dfb | |||
| ed21875083 | |||
| 0249b5b311 | |||
| f46a0f64f2 | |||
| 1fd0447dbb | |||
| 9856201f5c | |||
| a75c07065f | |||
| 6a86ef5a7d | |||
| 500e5d6759 | |||
| e3f2c9f29a | |||
| 3458e6248b | |||
| 14039d05a9 | |||
| c592da4937 | |||
| 61fb4244f2 | |||
| a503e022d8 | |||
| 0942dc0764 | |||
| 0d6919dd99 | |||
| 3e8e797bc2 | |||
| 0a907ec694 | |||
| f1b4a54991 | |||
| 58e7106df1 | |||
| 7bf391c54a | |||
| 07a6e098c8 | |||
| e4897d6b54 | |||
| 23026d966b | |||
| 3cd98df30d | |||
| 001be23258 | |||
| c85b6682e0 | |||
| bb2bc98a22 | |||
| 43da4f7ccc | |||
| 37e6d1e9c7 | |||
| 1d9926698e | |||
| b04f64e02c | |||
| c6d710b14b | |||
| 2e07850622 | |||
| 1b6888281e | |||
| ce94411dff | |||
| 58a3098cd8 | |||
| 8a68163905 | |||
| 1f98e64b71 | |||
| a9480a768b | |||
| ed94514fc2 | |||
| 1f41cc2030 | |||
| 6455ec36ec | |||
| 676a3d068c | |||
| 5533f8d87b | |||
| 93af7b61ce | |||
| 364ad5529d | |||
| 1974a46fe0 | |||
| 5327ab9903 | |||
| 235d191bae | |||
| 4eb80ec75a | |||
| 2900459222 | |||
| 9088792b0e | |||
| b41fe7beab | |||
| 00f9bb8a07 | |||
| a90e1637cb | |||
| 4bad5c0241 | |||
| ce09d82fc7 | |||
| 5b926b8196 | |||
| c719ccdb48 | |||
| 9f1786aeb1 | |||
| 0bb83e06bc | |||
| 2ba67c2bc2 | |||
| cae91983ec | |||
| 3190ef0c1f | |||
| c7c1372833 | |||
| ba02cda55f | |||
| 512a94afaa | |||
| 6d978c383a | |||
| 938678f87f | |||
| 9dcf5fca5b | |||
| 88f0dc4726 | |||
| 8ae7910aba | |||
| b70e85fd10 | |||
| 99bc591366 | |||
| f4963cd1c5 | |||
| 4859cb8528 | |||
| c42b91980b | |||
| 8b3c8820e0 | |||
| 29bc78a486 | |||
| 30fe86ed32 | |||
| 85cefd5a00 | |||
| b5a328e0ca | |||
| 209a2e8fc3 | |||
| 8b978e6aea | |||
| b4df0e7c9e | |||
| 76aeda6b00 | |||
| 1a0ad6d9c3 | |||
| 6b219e3e25 | |||
| 1e90691013 | |||
| 3577c87c88 | |||
| baf6607e74 | |||
| a700848bae | |||
| 9bbfd5804e | |||
| b234d74f43 | |||
| cd49ff330d | |||
| 49faf7af1a | |||
| c892b83c93 | |||
| 985b52331a | |||
| 021dded9dd | |||
| 8fd0cdc732 | |||
| fe7a4d42d3 | |||
| e89d6353af | |||
| c5bb74d184 | |||
| 874349c928 | |||
| 157604b3a5 | |||
| b935544d65 | |||
| 03271df579 | |||
| 0633d3a07d | |||
| 045377a594 | |||
| b5c8030f19 | |||
| 7c2072789c | |||
| f75e856d2b | |||
| 44d689bc6e | |||
| 9e433c2f19 | |||
| 21b6279b74 | |||
| 4d89076bdc | |||
| 71e4ff7e03 | |||
| d2dfda6583 | |||
| 6a855f528b | |||
| fb93109c2d | |||
| bd190af7a3 | |||
| 44268b0c6b | |||
| 6451e5e7d1 | |||
| 2b3c4c68e4 | |||
| 7030cf2433 | |||
| 26d7881b80 | |||
| 21fe42b28c | |||
| d5ecf68d26 | |||
| 0ddff4ec7d | |||
| 283ac3191f | |||
| 4dd0c80dad | |||
| 3b53c6ca47 | |||
| 4e3b4809ea | |||
| a90b8fb449 | |||
| 1da509027e | |||
| 5b96e4761e | |||
| 311ea79238 | |||
| 98be2c91df | |||
| 2657e5e226 | |||
| cfcb0d4fb7 | |||
| 97d03f3215 | |||
| 4065529bdf | |||
| 0a6260b1d8 | |||
| 12caf2510e | |||
| a58d2f710d | |||
| 6be2db8c42 | |||
| 5cf68416d8 | |||
| ebcb3c6b3b | |||
| 04267e0f6b | |||
| a540e6afc5 | |||
| f1841e48b3 | |||
| 9865bb6904 | |||
| e06ddea784 | |||
| 8b5a89c136 | |||
| da093c1982 | |||
| 048fb6278a | |||
| f1b0778f79 | |||
| 0e584fa4a5 | |||
| f44386008c | |||
| 60c139a844 | |||
| f410213003 | |||
| 19cb5d57db | |||
| 30b912fc81 | |||
| 7fc07e2d5e | |||
| 72c83d9430 | |||
| bfbac12f76 | |||
| 461f7dc9f9 | |||
| 3e5497e2f9 | |||
| 6ecbcc7c19 | |||
| 8cef02e8e8 | |||
| caabfd14b3 | |||
| 2ffbaa9578 | |||
| c40aeaec3a | |||
| 80e84a3ad0 | |||
| 0552335ec1 | |||
| 7aea774b21 | |||
| 4d4ed92055 | |||
| f0ec26992a | |||
| 62e8332b34 | |||
| 0925f71987 | |||
| 4c11652808 | |||
| 5b10c38e43 | |||
| 3316df9195 | |||
| 5d355f1a8b | |||
| 2ff91103ca | |||
| a954d50ad4 | |||
| 5be4d37aff | |||
| 1e6c9dbcfa | |||
| 708a56872d | |||
| 0a2bca3f73 | |||
| 1ec710c985 | |||
| d5a44f9ad4 | |||
| e64dca7144 | |||
| b2779c35df | |||
| 9b11e119d4 | |||
| 988c62baed | |||
| eb3e640003 | |||
| 9475b947f5 | |||
| 18564f1ae2 | |||
| 638f1deb62 | |||
| 7e74d30f45 | |||
| ce8d0f8135 | |||
| 872127b722 | |||
| e180dc44bc | |||
| 268b8845a9 | |||
| 74c47995a3 | |||
| 24f5936cbf | |||
| bdfa8aca28 | |||
| 15eb1ad922 | |||
| 6ec98ee8b1 | |||
| c5862d6de9 | |||
| 57eb55446f | |||
| abd1399a7f | |||
| 57eba21ee5 | |||
| 1b56211a70 | |||
| d8974d53b2 | |||
| 5ecd17f49e | |||
| 6bb99aec3c | |||
| a67e83e24e | |||
| 5a3c3134ec | |||
| ce8b9ee8c4 | |||
| 970dfc9f67 | |||
| 04d39c0961 | |||
| d6339aa015 | |||
| 11076bf337 | |||
| 8a8eea53a2 | |||
| 09bd7e8ef8 | |||
| 9356619380 | |||
| fc15147cf5 | |||
| ab6b7a8044 | |||
| 07c2fe726e | |||
| fa355603fb | |||
| 109bb505d8 | |||
| 4a6eebc0e4 | |||
| 78ce2b473e | |||
| 6beb5f5587 | |||
| 501fed6c4f | |||
| b1bd8e9ee2 | |||
| daeca1bb18 | |||
| 17f6d5208f | |||
| bee4d7a12b | |||
| f32e3e0c7c | |||
| 1b69612246 | |||
| 9e93509a56 | |||
| ef45cd3342 | |||
| 46fe2e6b44 | |||
| 1133c2cc1d | |||
| fde10553e0 | |||
| 83615ff351 | |||
| 352eb4cb6d | |||
| 75c75ac00c | |||
| d9bcf52db2 | |||
| 43f0362e6d | |||
| ed5e313c73 | |||
| c0010f60e6 | |||
| 75301e4cf5 | |||
| 875c8fdcbe | |||
| a0b1642dc0 | |||
| 1d7e54f8c9 | |||
| 77c8581bc0 | |||
| 5dd625916b | |||
| e3d7718cf3 | |||
| 9eccd7b1fb | |||
| 5b05d126b4 | |||
| 03aaf189c1 | |||
| 6ef9395419 | |||
| 3a56e13b78 | |||
| ec28acba3d | |||
| ee6647ce40 | |||
| 03d54f8f6e | |||
| 553e6d7549 | |||
| e6896ee71e | |||
| e6762f9b48 | |||
| 099bb1afef | |||
| 9c33093c91 | |||
| 634d8038b9 | |||
| ad46154f2f | |||
| c7fbb4615c | |||
| fa81068ea8 | |||
| 70c2a1c9f9 | |||
| 164fcb49d9 | |||
| 64cf18aa1e | |||
| 66a68ce264 | |||
| 86162aaddb | |||
| 9cc7a94a94 | |||
| 6bca1225e6 | |||
| 379a4e6a01 | |||
| 460cfcaf3e | |||
| 8e69103822 | |||
| 2f67dab2b6 | |||
| bb65ebd8be | |||
| c46ea0390c | |||
| bc8a6dd2e3 | |||
| b1478c37f6 | |||
| 4e944a9f3c | |||
| c7fa9b5fe8 | |||
| 65148b123b | |||
| 2f92a34bb7 | |||
| 54ed24f481 | |||
| 268df9f67a | |||
| 84dc398d32 | |||
| f6a3205d10 | |||
| 522cb66582 | |||
| f873a140ce | |||
| 36dfc5bbd1 | |||
| f80668e87f | |||
| dcb5d47ee6 | |||
| e95c22eb21 | |||
| 9fb83e61ea | |||
| 1513cdf7bc | |||
| e33af1a3f8 | |||
| 857d77a10a | |||
| 0e431a0250 | |||
| 3acfc0b630 | |||
| 2ce5f69def | |||
| 7d347be902 | |||
| a456d78fe0 | |||
| bf67c967d6 | |||
| 44b7a7145c | |||
| 3867ee71ed | |||
| 464f4813e3 | |||
| af8b52e7e8 | |||
| 796588900c | |||
| 4beb2ed507 | |||
| 0ff6833e96 | |||
| e2cfcc52b3 | |||
| fc8a46025e | |||
| d2bea0c228 | |||
| f016c2b72f | |||
| be62058696 | |||
| af18d5ed81 | |||
| e9c91a1ce2 | |||
| 29767b2886 | |||
| 96a31c69c5 | |||
| 534632dc52 | |||
| 8bf5f3d869 | |||
| c4f92322f5 | |||
| 1e4aa116e5 | |||
| 90cc1411da | |||
| d13ce6768c | |||
| fd4a7f2150 | |||
| 6bd64c6873 | |||
| ba58d868e5 | |||
| c50799ba3b | |||
| 039d82ff1b | |||
| a2f0933d01 | |||
| 77e1e3cc18 | |||
| 7bdd41350a | |||
| 6797a6ab56 | |||
| 86b5928f5e | |||
| 6dbd15aa71 | |||
| 22e5b081c4 | |||
| d848f33c48 | |||
| c64367536d | |||
| fc0102b079 | |||
| 62a39639c2 | |||
| 158aaff384 | |||
| fd836145fe | |||
| 697bafdd0a | |||
| 9675dcac44 | |||
| 48849d7866 | |||
| d0ce2f0b5a | |||
| a19f635a6a | |||
| 676ed59342 | |||
| 4015f46b7d | |||
| 770cee7139 | |||
| a4619a54a7 | |||
| f7d99f97a3 | |||
| 8b7df0c12e | |||
| bd780817f7 | |||
| 82fb45aa2a | |||
| 74870a8189 | |||
| 7a9f6b48f4 | |||
| d3c089130d | |||
| e0f3060527 | |||
| 4c1256acc4 | |||
| 85f6f5bd29 | |||
| 4d9eac663a | |||
| 0ef4d90ad0 | |||
| 1a1e7edb02 | |||
| 51b835f71b | |||
| e38fe3d361 | |||
| cc042c9936 | |||
| 599e3bc937 | |||
| 1fa0d940bc | |||
| 7dc4a9525b | |||
| b6f1f4ef64 | |||
| 3faae67663 | |||
| ccc94c9b05 | |||
| 1fd30db726 | |||
| 0ba76ac066 | |||
| 8b661fe556 | |||
| 077907b7c3 | |||
| 172d669780 | |||
| 6b85b9a416 | |||
| 5d3001279c | |||
| 4582a13360 | |||
| 3a064535ae | |||
| 444ec4ad27 | |||
| 94e910586d | |||
| bb5ce007e6 | |||
| deaa74d378 | |||
| 13e1794e91 | |||
| 67b3595008 | |||
| 6c33f518a8 | |||
| 88da62ba09 | |||
| b6997a56df | |||
| 74178fd1c2 | |||
| 34c59bfa90 | |||
| 2956bce047 | |||
| 41f33ecbb9 | |||
| 86241e2871 | |||
| 1e32897d3e | |||
| 1b63a9a9b5 | |||
| 4c9f11b78a | |||
| 32348c2b0b | |||
| 5e690c5d04 | |||
| 4ec56484a1 | |||
| 8f2a5649fe | |||
| c3b25e12a5 | |||
| 6d3e33d440 | |||
| c11f7ce54f | |||
| 84806cc174 | |||
| 16f12f7d59 | |||
| 8609b8e589 | |||
| 29e744fdbb | |||
| 515b87bcbe | |||
| 5fa9faca20 | |||
| a4ebdfd47c | |||
| 188d8d4b64 | |||
| ce5581d428 | |||
| 5b4acf14ea | |||
| 5fc6cb15b8 | |||
| f6e9a8eee4 | |||
| 6c0950cb2e | |||
| cb8a9ef2c0 | |||
| 470cbbe9ff | |||
| e01f1434fb | |||
| 187084ce46 | |||
| b6f9382b5b | |||
| 57f68381ed | |||
| 3e35729eb6 | |||
| 544fa57641 | |||
| 7e94309046 | |||
| 21eff5b825 | |||
| b408d7c95e | |||
| d9929edbc1 | |||
| 843b73dedb | |||
| 02a8145b18 | |||
| cfcb315b14 | |||
| c8a70a0a73 | |||
| b84a3a0230 | |||
| 49d70232f8 | |||
| 8cc9f496ee | |||
| 1547f2ec80 | |||
| 42a8b40de0 | |||
| b4b968ff44 | |||
| 52e3c063c5 | |||
| 257089884f | |||
| c650ea9765 | |||
| e369d45b9c | |||
| 2d84b6f6d9 | |||
| eef1171944 | |||
| 12ccdcf858 | |||
| f1a03bfb04 | |||
| 696b0e29e4 | |||
| 5eb748ae17 | |||
| aa4340ef5c | |||
| 0062e54e93 | |||
| dada5090b0 | |||
| 1c4593c648 | |||
| e7004cef76 | |||
| 2bb101bd19 | |||
| 26baf70912 | |||
| 89c2582376 | |||
| 33e003616d | |||
| bf03d77ab9 | |||
| a6cbf1f922 | |||
| d6f056f266 | |||
| 69a247d500 | |||
| a76c67c19f | |||
| b836164a38 | |||
| 058507badf | |||
| ad40e90790 | |||
| 0c9dc11550 | |||
| 1ff55c2729 | |||
| 066269153e | |||
| 38bb08778a | |||
| 5dbcdf1484 | |||
| f03a6ab5a4 | |||
| 6fa5abcd7e | |||
| 5dc07ed295 | |||
| 064d4255d5 | |||
| 04139eb82e | |||
| 1b1a122b1f | |||
| ccb132320c | |||
| c0e7f824df | |||
| 05bba71eaf | |||
| 7ebe5c4bcf | |||
| ae1bd891e7 | |||
| 9899e5021d | |||
| 94440e0170 | |||
| c25928e44f | |||
| a7fc7d4ffb | |||
| f336103f63 | |||
| 56e2b38048 | |||
| a5ccff720a | |||
| 1d8c2d6c22 | |||
| 0b8c357eff | |||
| efc168f473 | |||
| d8428f98d9 | |||
| 60f17d26a3 | |||
| 05bc664c11 | |||
| 2cc84b6e51 | |||
| 5ccdbef7d5 | |||
| c13c2650a2 | |||
| ec6c998a3a | |||
| 2f6091419f | |||
| 2022dd7d74 | |||
| b8202dab3b | |||
| ef688a74fe | |||
| f632e7c043 | |||
| 04a19f9813 | |||
| 2cbc591c9d | |||
| 3f00e79bcb | |||
| 3586fc4910 | |||
| c9a6bbeb64 | |||
| 0655a135e6 | |||
| d3e8bb1889 | |||
| 14ceacac73 | |||
| e4f33b5970 | |||
| 4474f8ef18 | |||
| 76c9f4f5a6 | |||
| 942ef3b7f2 | |||
| 0b9df6d8c4 | |||
| b5ea504ad2 | |||
| 803b0c4bdb | |||
| 6537d0dc76 | |||
| f8f36c085c | |||
| 7339f67dd7 | |||
| 0d4e501239 | |||
| 8d609607e2 | |||
| 71a889ed73 | |||
| 27a75a9085 | |||
| 954d6c326d | |||
| 7ea05d038e | |||
| 33930ff046 | |||
| 16f41ea059 | |||
| 0a7270fc29 | |||
| 23fbd9d004 | |||
| 610c79fbf3 | |||
| fd44c2a2ff | |||
| a86a82b39c | |||
| 89b059b1ea | |||
| bd2d0f769f | |||
| d830422489 | |||
| d1a54249e7 | |||
| 4dfbf98e4e | |||
| 1b6258ec8c | |||
| be707dbb6f | |||
| 664b03bb13 | |||
| 7c6723d912 | |||
| 1febf2ec83 | |||
| fe69928764 | |||
| bbd61eb13f | |||
| e15e1e253d | |||
| 45e2178ada | |||
| a6e4933d93 | |||
| 98599e0972 | |||
| b4837f2e2f | |||
| ea08e7d192 | |||
| d178e089a6 | |||
| 5f00b37e21 | |||
| 8a8792d47f | |||
| 59d9bc9e48 | |||
| 8793dd3ceb | |||
| 48062380fa | |||
| 3636aa5522 | |||
| a1aea4588f | |||
| 1d4fffb799 | |||
| 6f90f5dc5f | |||
| 9dd6972d26 | |||
| d731a7d52c | |||
| 059468b74e | |||
| 5e69fb782a | |||
| a5beffda78 | |||
| 7de7ce5fdc | |||
| 383e8c7f68 | |||
| 0dbda65e44 | |||
| fe01da077e | |||
| d43a4e9df9 | |||
| 3e226795f0 | |||
| c4a0fe1606 | |||
| ef63a84a3e | |||
| 8c16ba372e | |||
| 9be4a17687 | |||
| 89332e1696 | |||
| 8a56129def | |||
| ed0c815735 | |||
| 7a69da16e4 | |||
| 351717414d | |||
| 52f44de257 | |||
| ae6dddfff4 | |||
| 539f555e23 | |||
| b75fa26dc1 | |||
| 3d22a2d845 | |||
| 1aab4752e2 | |||
| 86f8a4a9d2 | |||
| db2cb061cb | |||
| 6a71b24495 | |||
| 84712a8bbc | |||
| b86fb95306 | |||
| a3a9bde83e | |||
| 2fe2dd170b | |||
| f772bf4fbc | |||
| ac0c3093f4 | |||
| 12150baa5e | |||
| 219b02c1e5 | |||
| 4551e60f8b | |||
| ea842e78af | |||
| 5651fbedc4 | |||
| 2f87dfd4a9 | |||
| 332dd76867 | |||
| d2c9ea8a9a | |||
| 40d57da83c | |||
| 561813eb2a | |||
| e6c9dfbd91 | |||
| 603b6596af | |||
| 64abc3e86c | |||
| fac42fa3e8 | |||
| 7ad4020829 | |||
| 72ab0d11ff | |||
| 4ea866f050 | |||
| a476531524 | |||
| e7e6ac5bb3 | |||
| f346362b00 | |||
| d9ba455695 | |||
| 8927a0561f | |||
| e03c5e9f23 | |||
| 1f79200db8 | |||
| c009e4a57d | |||
| c615d52cf4 | |||
| dbb3316511 | |||
| cd6f204c77 | |||
| 269131ed21 | |||
| 65d784e88e | |||
| 35afb6cae0 | |||
| 27bce09be8 | |||
| 4f25b6ac0c | |||
| 3d5ed1a7e3 | |||
| a03115a4a6 | |||
| 7219d28a31 | |||
| 0875bce68f | |||
| 54fe302907 | |||
| edaa8f811f | |||
| 2c8fd109de | |||
| 07fe7ad1a2 | |||
| 16d88cc095 | |||
| 2a6e6b3dbd | |||
| 0c19848230 | |||
| 3c3a4db54e | |||
| 0e6bd2224f | |||
| a64d2f4673 | |||
| 1e8a54af0b | |||
| aa53d8708e | |||
| 8c600ca553 | |||
| 25fe6d7dde | |||
| afb369950c | |||
| d7f133c24c | |||
| 5312fd30e5 | |||
| 23dd0bdaa1 | |||
| 1d06624d38 | |||
| d40069a018 | |||
| 73e27bdd48 | |||
| 064eb0b24f | |||
| af968c5b44 | |||
| f93dbe51e2 | |||
| 1c34707925 | |||
| deaca58504 | |||
| 1d519aa9cb | |||
| 191faeae70 | |||
| 1153aaf55b | |||
| 292cb5a5af | |||
| 1e9488d4a6 | |||
| ff1d77ead9 | |||
| 977e1a94b2 | |||
| 60ee5fc844 | |||
| 940c8fcb5e | |||
| 71e0148eb4 | |||
| 293c104cc4 | |||
| 5a3035bb72 | |||
| e04cbd71d0 | |||
| fa4ce6a8bc | |||
| 9863f62321 | |||
| 7cd1f7dbd5 | |||
| 8c45a18524 | |||
| 8cb383ed45 | |||
| 7d1305b169 | |||
| 24ab1f32e4 | |||
| 073ad0dada | |||
| f43b6a0675 | |||
| fc1ddcd2f8 | |||
| e7f774f964 | |||
| cb49af1ea5 | |||
| 73d7d704c1 | |||
| 8b89232f12 | |||
| c3dec1a5ea | |||
| 44b06d70e8 | |||
| eee07e6cfd | |||
| 5051c27c3d | |||
| 76f3506ac5 | |||
| 5d7a84fad7 | |||
| dec161ed26 | |||
| 0f9dbf84b7 | |||
| f0d5337818 | |||
| 4cd9de5c37 | |||
| 76b0bfa7f5 | |||
| 32d6b0eed4 | |||
| 127d962271 | |||
| 92d7af0881 | |||
| af12066f77 | |||
| 2a1f8fa8f1 | |||
| 04e47bde84 | |||
| 49da7e74cd | |||
| 6ac47734c0 | |||
| 0e6ea76e88 | |||
| 0514588175 | |||
| 59d1212039 | |||
| d61cca6720 | |||
| 8c74e88f16 | |||
| 46cf512032 | |||
| 24a185d26b | |||
| b99a7344c9 | |||
| c6a4fb1e13 | |||
| e2718fe845 | |||
| 76314280cb | |||
| d6716218bd | |||
| 9371a92122 | |||
| e8b030ad17 | |||
| e0180b4849 | |||
| 3901bbb401 | |||
| 76bebfd798 | |||
| 3013166d8d | |||
| 8596e702ac | |||
| 414bf4a296 | |||
| 0daa01edef | |||
| 98abd96075 | |||
| 8e3fc826e2 | |||
| c750095241 | |||
| 2a0c0c0ad2 | |||
| 92f3bb89c3 | |||
| ac0e6c5e6e | |||
| f397b6fedf | |||
| 1d069e5077 | |||
| c5684a6278 | |||
| f3ac0be0e6 | |||
| 18c9468af5 | |||
| 32bc0da362 | |||
| c9a3800ce7 | |||
| d4239aaa8f | |||
| 4f72d5cfac | |||
| 66acab4130 | |||
| 9e9e3373e0 | |||
| 502fee1b45 | |||
| 4d0c7d706d | |||
| 6cd418b60a | |||
| a3b39dfd1a | |||
| 74da47e286 | |||
| 587ba9bec0 | |||
| 382392e03b | |||
| 409948a0f9 | |||
| 34919ca394 | |||
| 0d1c574cb1 | |||
| c564815931 | |||
| f6fb667ac1 | |||
| 832bbe734d | |||
| 7a2fda891c | |||
| e50c239a2e | |||
| cc7c8d92da | |||
| 87acab0846 | |||
| f43459d476 | |||
| e030f02776 | |||
| 10f2d01e7f | |||
| 80dbf9a32a | |||
| 185274e70f | |||
| f0ac55ec0c | |||
| 44544635dd | |||
| 3c594b1037 | |||
| ea7100e8c4 | |||
| a198abc485 | |||
| d4a37f6ef5 | |||
| a5c9c31231 | |||
| e3ec78a832 | |||
| a116e68a47 | |||
| e7084de166 | |||
| 398eda6365 | |||
| 349abf5ee6 | |||
| 4ce40b1975 | |||
| 536fe28f8f | |||
| 3e9e14f4d6 | |||
| c8140068ad | |||
| db314bc381 | |||
| 499a26b152 | |||
| a9cdb5be50 | |||
| 5f04208dbd | |||
| ffaa292006 | |||
| d3e44b1108 | |||
| fbf274a42b | |||
| d94cd65dfd | |||
| 3c1b403c4e | |||
| 75564453b3 | |||
| 3091e2dc0e | |||
| fc50a36cc5 | |||
| 9bf9fba2ec | |||
| 121615da70 | |||
| 53d28a713c | |||
| cf37704193 | |||
| 38289fe381 | |||
| f9337a1111 | |||
| 6d059b479f | |||
| b6e23b2d3e | |||
| e5e6a46c37 | |||
| 22b9a53bef | |||
| 3be81e3206 | |||
| ff09b6c824 | |||
| a8e892ba90 | |||
| 289cc3e7a0 | |||
| 7480b87e07 | |||
| befa6423be | |||
| fd418f568c | |||
| 09cf18a646 | |||
| 326c175dcb | |||
| 6d7c77ddc1 | |||
| efd706528b | |||
| b523c43927 | |||
| 75545ff70d | |||
| 3c6ef83046 | |||
| b9ac0a79f1 | |||
| 8539896f3d | |||
| a3b508ceff | |||
| 334a486737 | |||
| d7370cc916 | |||
| 92c34f7f38 | |||
| 93328c8d6d | |||
| 5710ec13d4 | |||
| fa637fcecb | |||
| 4af7d6f108 | |||
| 0fd159dadb | |||
| 1ff22c78b3 | |||
| ceb1def55c | |||
| 893a1d8306 | |||
| 3c91690e55 | |||
| 6835dd73bc | |||
| 7317fe1440 | |||
| 7b58fea911 | |||
| c1ff74c9a6 | |||
| 6dabfa176a | |||
| 0714f5fc67 | |||
| 3b1b1bfd48 | |||
| 218c867f46 | |||
| 2bc12f9730 | |||
| 5e564a8e0c | |||
| beaa6a9a7a | |||
| a9c8224f40 | |||
| 3dcc188d93 | |||
| 10b7556a37 | |||
| 54b7291c34 | |||
| 1e30b6e334 | |||
| 406240bae3 | |||
| 74d9b41b7d | |||
| ff0b0c54b7 | |||
| 6eec2d6b4f | |||
| 5731c5437a | |||
| 04f14ec026 | |||
| 12ed6336b1 | |||
| 3cb79e6977 | |||
| c5e21a2469 | |||
| 13aee51011 | |||
| 53fca1b5e6 | |||
| 7dad9fca0f | |||
| b249d7c76c | |||
| 4060f64232 | |||
| 5b2f7d3374 | |||
| 3116e29d16 | |||
| d2406f2a22 | |||
| a648318900 | |||
| 849f54e4f8 | |||
| 61009fea3f | |||
| 21dce6cca9 | |||
| 56bc8a778d | |||
| d93af1161d | |||
| 434776db1a | |||
| 6167e9cefc | |||
| dc918d764e | |||
| 9906887151 | |||
| b5a1017afa | |||
| 7badc230a4 | |||
| ab78482ee7 | |||
| 6369cf4dd9 | |||
| 7656bd50ee | |||
| 2115596ed3 | |||
| 0e3453f7c2 | |||
| 7ed65e42d7 | |||
| 835b640ebd | |||
| 6ee3318531 | |||
| ae24fe3850 | |||
| d4f4608dab | |||
| ea8a5020e2 | |||
| 622d9c9480 | |||
| bb0c9547be | |||
| 62da98aef6 | |||
| 03746b966b | |||
| de001da35b | |||
| 0da460ca13 | |||
| 450e19858b | |||
| 836e1fc330 | |||
| e836c28008 | |||
| fff4f921e4 | |||
| 9f265711a8 | |||
| 32afcd2e48 | |||
| 748df8d109 | |||
| 5ad405006c | |||
| 47859f3560 | |||
| 56d1b9a226 | |||
| 1b6a31b277 | |||
| 90a7503181 | |||
| e3efbcddc1 | |||
| c14b2fb36c | |||
| f0f111b387 | |||
| c95e45d283 | |||
| 52b8d50178 | |||
| f581627d10 | |||
| 6a8ec95a46 | |||
| b6c6680add | |||
| 5fb149f833 | |||
| abb0bf9247 | |||
| 8f3ddd3a73 | |||
| 8f34e6714a | |||
| 56841bcede | |||
| 006cc2ed60 | |||
| c79cf8d6bf | |||
| cf704cf81b | |||
| 265d474ec8 | |||
| 2e420169c3 | |||
| 5aec2671ea | |||
| ab0e22a316 | |||
| 06587824be | |||
| a0bce440a6 | |||
| 26b15251e2 | |||
| 1cf4fe405d | |||
| 6b8f5d3354 | |||
| d5af359365 | |||
| 8769e42a56 | |||
| 7cde65aa6e | |||
| e1b1500e3b | |||
| 2943a1c27f | |||
| b28cafc1d1 | |||
| 65f999b7b7 | |||
| 06d6636b97 | |||
| eb5a1ea113 | |||
| d84e70b6e5 | |||
| 7ff034504d | |||
| dedf0c6a8d | |||
| 5514ae3879 | |||
| 5af0dfb031 | |||
| 0bcda5e384 | |||
| d1eef242c6 | |||
| ceee00b276 | |||
| 6c2ab064cb | |||
| 772a5dc3d5 | |||
| 3e39a998ce | |||
| 2867dc50fa | |||
| c3c43769ae | |||
| 36ceaa4452 | |||
| c34b1a1b2a | |||
| e4df0ca368 | |||
| 8a91cecf41 | |||
| 04e8710cf5 | |||
| e8b3f9eaad | |||
| 23d6ec6cff | |||
| 0a6edae2dd | |||
| 0e45663ce8 | |||
| 80a9f4defd | |||
| 8f1e4018c0 | |||
| afe36d0b36 | |||
| 5d1e3efce8 | |||
| 293ec7aec5 | |||
| 6cefeb338b | |||
| f1744f5495 | |||
| 5750a173ff | |||
| e3a4fd9f93 | |||
| 5a071c1907 | |||
| 7cf3a7511b | |||
| af203aaf86 | |||
| 8e2c06cb0e | |||
| 1a5d8f1957 | |||
| 03c828c7ad | |||
| 758dc511fb | |||
| 0164723a8e | |||
| 032936a7b5 | |||
| da3e064fc7 | |||
| 317fc6ba0e | |||
| 1aaad223c0 | |||
| 81c86d7090 | |||
| d9a9fd387d | |||
| 0c190b165c | |||
| acc7bd79b0 | |||
| e4e89fe27a | |||
| 24551db0c8 | |||
| 12e6611ba4 | |||
| fb15886a1c | |||
| 06c1dc3a29 | |||
| 89d9de2353 | |||
| 12c85d3e23 | |||
| a5afec1f94 | |||
| ac0899c043 | |||
| 40c6213d7e | |||
| d140bc23f5 | |||
| 00f0859e1f | |||
| a0b2fab6fa | |||
| f669aafcf2 | |||
| 66a2807210 | |||
| c3009eb324 | |||
| 3bdfe167de | |||
| 31e8a12e88 | |||
| ebbfdcd35a | |||
| 9a7c8fb5be | |||
| cfef4ff2ad | |||
| b2220d6157 | |||
| 5ff941ae3d | |||
| 1c922d3b73 | |||
| b23dd28a06 | |||
| a55f41a24a | |||
| 5525c6f729 | |||
| eb147d9868 | |||
| f58a5d534e | |||
| b3ea8c406e | |||
| 99667f7c55 | |||
| 0b21203141 | |||
| 9a9ca974c2 | |||
| 140e4dde3d | |||
| 68670301e3 | |||
| a98d841983 | |||
| 332b764cc4 | |||
| 560f0742cc | |||
| 910f272467 | |||
| b6423a3426 | |||
| 4d2736ffa9 | |||
| 4dc2adf7f8 | |||
| da34f9a253 | |||
| 1f76737510 | |||
| bc8bc7d1a8 | |||
| 083569fca8 | |||
| a8903d9765 | |||
| 8e7d1a5f09 | |||
| c879b56f41 | |||
| 4518f1fba1 | |||
| 5c59b3a775 | |||
| 299dfcdd3c | |||
| 76c706644a | |||
| 0c8f2b9d85 | |||
| c924aaede9 | |||
| 28710f8ad5 | |||
| e695a19d11 | |||
| 6978a0b8d4 | |||
| 6784530b8b | |||
| 03d5fc33ca | |||
| ba14232628 | |||
| 1cdf5581f3 | |||
| 3488c49d0a | |||
| adaef43bc6 | |||
| ce8fe1bdf6 | |||
| fa04595d90 | |||
| aea79912ec | |||
| 80b4dd2e8a | |||
| 48530b89ea | |||
| a4025788ae | |||
| c6f2f60b03 | |||
| 33060738b6 | |||
| ab6d4871d8 | |||
| f87e64f988 | |||
| 27861f6358 | |||
| f611b65bc0 | |||
| 2dc61fbdc4 | |||
| 22be05400d | |||
| a9f501fe7d | |||
| e9077370ec | |||
| a804351a76 | |||
| f97b655f02 | |||
| 1498b78342 | |||
| 85e84fc1fa | |||
| 833e5d8bf1 | |||
| 773883c486 | |||
| 6e5e0278c2 | |||
| 951c4bedf8 | |||
| 9842e1f9d0 | |||
| 4c0c1c9830 | |||
| 6706d6053e | |||
| 0a874a5063 | |||
| 2caa6e3370 | |||
| a9e990251d | |||
| 7bde23590a | |||
| 5042dd52ce | |||
| 3b9e6bff3c | |||
| a2d05b21ff | |||
| f4f5f670a2 | |||
| 165e23773f | |||
| 8dbb598057 | |||
| ba9dc12164 | |||
| 399d08c86c | |||
| 6f799435b6 | |||
| 3d14154a29 | |||
| 7e331957c4 | |||
| 4da06830f1 | |||
| 27293cc1c1 | |||
| 2caac2b218 | |||
| 0dc80ccf21 | |||
| f2b48ede4c | |||
| 6cefdc2f5c | |||
| 29e78413fe | |||
| 8192e63a4b | |||
| b2ebdb0d07 | |||
| 9c3828fefe | |||
| 60916318f7 | |||
| d7c83397e4 | |||
| 1d621bba37 | |||
| e2f349e7bd | |||
| 102262c7ab | |||
| f02babe427 | |||
| fc6133b58f | |||
| 2bd65fa444 | |||
| d33208c7db | |||
| c9cd8e6211 | |||
| 74a96878bc | |||
| 7e28708e1d | |||
| 1211c01ca1 | |||
| f32b97733b | |||
| 4e1c90f76f | |||
| e63f258470 | |||
| ede9f9117f | |||
| 7c560fa137 | |||
| 178a0842fe | |||
| f345490cae | |||
| db141e82c9 | |||
| f163155929 | |||
| 9b6377fd80 | |||
| 7356b4532f | |||
| 7d7bec856d | |||
| c5504ef50b | |||
| 3658ff650d | |||
| 6d14afd80e | |||
| 6e5178efc4 | |||
| 29fc51522a | |||
| 6cd8fb7982 | |||
| ce824f8653 | |||
| 708f4a094d | |||
| 2704b73399 | |||
| 783ccd6c21 | |||
| 3fd1c3b64a | |||
| 58d249ca16 | |||
| bdc2b07339 | |||
| 6888ca709d | |||
| 8ae818e17c | |||
| c4f1baad31 | |||
| 3439ce19c9 | |||
| b7c18df540 | |||
| 74799134b1 | |||
| 3828e1e538 | |||
| d89046d515 | |||
| 4bc128f07e | |||
| e383b7a6ab | |||
| c89d6bf68b | |||
| 52640518d3 | |||
| cf493254b7 | |||
| c97eb41dc6 | |||
| b1224a77db | |||
| 17b777f751 | |||
| 9442c9e1f4 | |||
| 15740500af | |||
| 3484dda45e | |||
| 1ece6c0e2f | |||
| 59cad23aeb | |||
| da1c35d04b | |||
| c469aed047 | |||
| a065805b0f | |||
| f41a18b57d | |||
| 1257432df3 | |||
| d0c0e31220 | |||
| 9b7832c39a | |||
| e3e29b720d | |||
| 64872bddf4 | |||
| 81f2249575 | |||
| 9bbd6bd874 | |||
| 13e477ebfe | |||
| c284b54fc2 | |||
| 69caa477fb | |||
| 81f9aac13f | |||
| b2eff3c90c | |||
| aa45fe7359 | |||
| 253af0766c | |||
| cf9dbe583d | |||
| de8df0a05f | |||
| 53b6deaeae | |||
| 462858efa3 | |||
| f7e893667d | |||
| 5765c81f66 | |||
| 92334a8e28 | |||
| c4218c8e40 | |||
| c1f27fb848 | |||
| 6d0fd5bb93 | |||
| bd15d3ae24 | |||
| 7f249cd179 | |||
| aef3f4be99 | |||
| bf8083888d | |||
| f4fa5b7340 | |||
| 169568ca47 | |||
| 9cc4ddfc88 | |||
| 441963c84c | |||
| cf4ae61ac6 | |||
| da0f1cacea | |||
| 5e5592178d | |||
| b01222518d | |||
| 2060cf8a70 | |||
| a4bd87119b | |||
| 9f26355fe0 | |||
| f667d4965d | |||
| 585f84a734 | |||
| a1bff85263 | |||
| fb920bba62 | |||
| 58697f6f3b | |||
| 18c5b8d68a | |||
| 08cf140811 | |||
| 94673bcdf2 | |||
| aa15917c9d | |||
| 85fb37b6ea | |||
| ae3ae9a474 | |||
| e9be643db5 | |||
| b49eefbee6 | |||
| 640283fec6 | |||
| c8d50a6060 | |||
| 6a2728e730 | |||
| 1740d93420 | |||
| 0042d9b406 | |||
| 3bfa6097d5 | |||
| 237b8865f5 | |||
| 8f01cece3a | |||
| 2ca574d9e6 | |||
| 7f27e1e0e1 | |||
| 875e2f9d0d | |||
| 5c538dd9d6 | |||
| 3fb82502f7 | |||
| 7be2998cae | |||
| 9dfab9d9a4 | |||
| b63ae1f190 | |||
| 4fc796b387 | |||
| c3ff41dd84 | |||
| 0b927f059c | |||
| f3c3afd4cd | |||
| 1e26859bb7 | |||
| b1beacd1f3 | |||
| d9a0e2b8f4 | |||
| 4c7e95aac3 | |||
| c3310c6e8f | |||
| 2a24567370 | |||
| bd9628df93 | |||
| 99a153d9e8 | |||
| 04da71c3a1 | |||
| 0c86dd9d8d | |||
| 349068dcda | |||
| 144b10b35d | |||
| 44722bddcf | |||
| 2a240e3fe2 | |||
| ee66fb1c60 | |||
| ea212e4b50 | |||
| 038b18edf1 | |||
| 0610ebc514 | |||
| 968117c940 | |||
| 6788b12d65 | |||
| 66ffc1b2d6 | |||
| 4c7d384e9a | |||
| d83aef4e86 | |||
| bf59ba76f5 | |||
| e8e78f8be6 | |||
| 76da659977 | |||
| c2eea8abba | |||
| 065805d6e1 | |||
| 5fa79b2db2 | |||
| 5f20d3eb34 | |||
| 3c0f5a3fe4 | |||
| 771e9cd68a | |||
| c328afee57 | |||
| 3dae86223d | |||
| dd07212f02 | |||
| 85e31a5479 | |||
| a53d95099c | |||
| f76ee5e5ef | |||
| eba02dc1b9 | |||
| bcabdfc1ae | |||
| cb44b3b9f2 | |||
| 9aa2cd71b2 | |||
| abdf81b39b | |||
| 1176725af7 | |||
| f668adcf11 | |||
| a3beac8d13 | |||
| e926b4b3c9 | |||
| 6c168f046d | |||
| 37fa6affc8 | |||
| 47c6490115 | |||
| 467831e3d8 | |||
| 3c1638c046 | |||
| 4b7e87ec7f | |||
| 98b387aac3 | |||
| 5312c7ff31 | |||
| be956654b2 | |||
| 977f57fd37 | |||
| a1ea37c336 | |||
| 7369339c88 | |||
| 8ace2ba194 | |||
| 3f79385160 | |||
| 14ee003907 | |||
| 3bd3116cf8 | |||
| a1f692408d | |||
| b0d9c074e1 | |||
| 9238e15bb1 | |||
| d7b9a29dc6 | |||
| 0c2f58e40c | |||
| fba27ef4b9 | |||
| 19cdc09928 | |||
| 2b2d93b05f | |||
| 36471c23ce | |||
| bc88aaf8d7 | |||
| f598a8666c | |||
| 85b5ccf5ae | |||
| a592199068 | |||
| f7ea2629e4 | |||
| 83a9fa8913 | |||
| 477b058f74 | |||
| 861a6a17e4 | |||
| 0df6d83f08 | |||
| 036f9d5a45 | |||
| 43143f6434 | |||
| 3f24879157 | |||
| 40c098f78a | |||
| b335af8507 | |||
| 9a6a146183 | |||
| 9230588ce8 | |||
| 78406ba954 | |||
| aa78b70d69 | |||
| 1b81e7c928 | |||
| de08df6a7e | |||
| c2e4b8ca9a | |||
| 6723221a42 | |||
| 5bd7fffb4c | |||
| 471c71310b | |||
| 9e79acc25a | |||
| cdb06a4c6e | |||
| e1af3737f3 | |||
| d7f7f1b200 | |||
| 8914b12db5 | |||
| 296777546c | |||
| 3db8c5a0eb | |||
| b0e6bfa84c | |||
| b1e8990654 | |||
| 463ef9b08f | |||
| 14016743be | |||
| 59194dcf4d | |||
| b32c72f1fc | |||
| e27a46973c | |||
| 06461a465b | |||
| cf6f231be6 | |||
| b1d5849bb5 | |||
| cdc75dec97 | |||
| 9239f75123 | |||
| f0bee2ac8b | |||
| 295e481a2e | |||
| 5aaca27cda | |||
| f220c1e9eb | |||
| 642132920f | |||
| 4e7e7d99cc | |||
| ba8aa46cd0 | |||
| f00be30318 | |||
| 6b5231f930 | |||
| 2c7a9734af | |||
| 8526387acb | |||
| 17ac5c0525 | |||
| bf82288ab1 | |||
| 2151ad7f34 | |||
| 576914ed54 | |||
| 43dba8ac7f | |||
| dcd0cb8080 | |||
| 47beaff152 | |||
| e4bae80f9b | |||
| 1d531a9600 | |||
| 14cd1f7a0b | |||
| 871fd20ee5 | |||
| 58f0d81925 | |||
| b98454d213 | |||
| 954b89e762 | |||
| f75280ac9c | |||
| e370a65383 | |||
| c5a3f9ccd4 | |||
| 20cda07eef | |||
| c1975166a0 | |||
| f0574d492c | |||
| d8fa44f17e | |||
| 9447828c3a | |||
| 39fcc62e85 | |||
| 6f0d350f2c | |||
| 719dff1312 | |||
| 7c8404eaf9 | |||
| 4c812d47bd | |||
| 0d81fd287e | |||
| 681cd33698 | |||
| 1153778f92 | |||
| 49332d3e90 | |||
| 4495619a2e | |||
| 33f45582af | |||
| c9c8e14684 | |||
| 5eb3610a01 | |||
| f4a06036c6 | |||
| d4c03ce6cf | |||
| 9368d7c25c | |||
| de497675ac | |||
| a66bd48cae | |||
| 0250352139 | |||
| 134ba8d1dd | |||
| 777b9c9a9e | |||
| 5ba29122fd | |||
| ddc2867f94 | |||
| b81310cb82 | |||
| b4c815a60c | |||
| 473ab12a0a | |||
| 2c23b375b2 | |||
| 5578401a0f | |||
| 9b6d32346b | |||
| 783132318f | |||
| 7f3aa316a8 | |||
| 440ef26b44 | |||
| b84c92be3d | |||
| 40a5d5ddfa | |||
| 374fe1af1e | |||
| bf9b1b1457 | |||
| d5a185b13e | |||
| 40ad3d6c92 | |||
| df8f792183 | |||
| 609e96b5d1 | |||
| bb95747d56 | |||
| afd9d72954 | |||
| 596d79cde2 | |||
| d9cc78615f | |||
| 55e9b082cb | |||
| 70f8fc1b42 | |||
| da970585ca | |||
| d2fa086198 | |||
| 5dc47ac0ea | |||
| f8566871a0 | |||
| c1dea29501 | |||
| 8fc77a4365 | |||
| 21eef55907 | |||
| d5a35c8fc2 | |||
| d873ee9983 | |||
| 6628c365c9 | |||
| 0ca170b130 | |||
| 5040840578 | |||
| 0213c3b4d6 | |||
| 2851ea490c | |||
| 6541682433 | |||
| 693b45c561 | |||
| 6102b310b2 | |||
| ec2b664d8e | |||
| cd82418ee7 | |||
| 352dd5e7fa | |||
| 10b6b0445e | |||
| b0c4302887 | |||
| db1702623c | |||
| 0c8ee105b4 | |||
| a22c20fab0 | |||
| 9ddc8d6ba6 | |||
| 3d48628e71 | |||
| 696e7175b7 | |||
| a78547835e |
+50
-11
@@ -1,17 +1,56 @@
|
||||
version: '{build}'
|
||||
branches:
|
||||
only:
|
||||
- master
|
||||
image:
|
||||
- Visual Studio 2017
|
||||
clone_folder: c:\projects\simdjson
|
||||
branches: { only: [ master ] }
|
||||
configuration: Release
|
||||
image: Visual Studio 2019
|
||||
platform: x64
|
||||
|
||||
platform:
|
||||
- x64
|
||||
cache:
|
||||
- C:\dependencies -> dependencies\CMakeLists.txt
|
||||
|
||||
environment:
|
||||
# Forward slash is used because this is used in CMake as is
|
||||
simdjson_DEPENDENCY_CACHE_DIR: C:/dependencies
|
||||
|
||||
matrix:
|
||||
- job_name: VS2019
|
||||
CMAKE_ARGS: -A %Platform%
|
||||
- job_name: VS2019ARM
|
||||
CMAKE_ARGS: -A ARM64 -DCMAKE_CROSSCOMPILING=1 -D SIMDJSON_GOOGLE_BENCHMARKS=OFF # Does Google Benchmark builds under VS ARM?
|
||||
- job_name: VS2017 (Static, No Threads)
|
||||
image: Visual Studio 2017
|
||||
CMAKE_ARGS: -A %Platform% -DSIMDJSON_BUILD_STATIC=ON -DSIMDJSON_ENABLE_THREADS=OFF
|
||||
CTEST_ARGS: -E checkperf
|
||||
- job_name: VS2019 (Win32)
|
||||
platform: Win32
|
||||
CMAKE_ARGS: -A %Platform% -DSIMDJSON_BUILD_STATIC=OFF -DSIMDJSON_ENABLE_THREADS=ON # This should be the default. Testing anyway.
|
||||
CTEST_ARGS: -E "checkperf|ondemand_basictests"
|
||||
- job_name: VS2019 (Win32, No Exceptions)
|
||||
platform: Win32
|
||||
CMAKE_ARGS: -A %Platform% -DSIMDJSON_BUILD_STATIC=OFF -DSIMDJSON_ENABLE_THREADS=ON -DSIMDJSON_EXCEPTIONS=OFF
|
||||
CTEST_ARGS: -E "checkperf|ondemand_basictests"
|
||||
- job_name: VS2015
|
||||
image: Visual Studio 2015
|
||||
CMAKE_ARGS: -A %Platform% -DSIMDJSON_BUILD_STATIC=ON -DSIMDJSON_ENABLE_THREADS=OFF
|
||||
CTEST_ARGS: -E checkperf
|
||||
|
||||
build_script:
|
||||
- mkdir build
|
||||
- cd build
|
||||
- ps: cmake -DCMAKE_GENERATOR_PLATFORM=x64 ..
|
||||
- cmake --build .
|
||||
- ctest --verbose
|
||||
- cmake --version
|
||||
- cmake %CMAKE_ARGS% --parallel ..
|
||||
- cmake -LH ..
|
||||
- cmake --build . --config %Configuration% --verbose --parallel
|
||||
|
||||
for:
|
||||
-
|
||||
matrix:
|
||||
except:
|
||||
- job_name: VS2019ARM
|
||||
|
||||
test_script:
|
||||
- ctest --output-on-failure -C %Configuration% --verbose %CTEST_ARGS% --parallel
|
||||
|
||||
clone_folder: c:\projects\simdjson
|
||||
|
||||
matrix:
|
||||
fast_finish: true
|
||||
|
||||
+278
-65
@@ -1,82 +1,295 @@
|
||||
version: 2
|
||||
jobs:
|
||||
"gcc":
|
||||
version: 2.1
|
||||
|
||||
|
||||
# We constantly run out of memory so please do not use parallelism (-j, -j4).
|
||||
|
||||
# Reusable image / compiler definitions
|
||||
executors:
|
||||
gcc8:
|
||||
docker:
|
||||
- image: ubuntu:18.04
|
||||
- image: conanio/gcc8
|
||||
environment:
|
||||
CXX: g++-7
|
||||
steps:
|
||||
- checkout
|
||||
CXX: g++-8
|
||||
CC: gcc-8
|
||||
BUILD_FLAGS:
|
||||
CTEST_FLAGS: --output-on-failure
|
||||
|
||||
- run: apt-get update -qq
|
||||
- run: >
|
||||
apt-get install -y
|
||||
build-essential
|
||||
cmake
|
||||
g++-7
|
||||
|
||||
- run:
|
||||
name: Building (gcc)
|
||||
command: make
|
||||
|
||||
- run:
|
||||
name: Running tests (gcc)
|
||||
command: make quiettest
|
||||
|
||||
- run:
|
||||
name: Building (gcc, cmake)
|
||||
command: |
|
||||
mkdir build
|
||||
cd build
|
||||
cmake ..
|
||||
make
|
||||
|
||||
- run:
|
||||
name: Running tests (gcc, cmake)
|
||||
command: |
|
||||
cd build
|
||||
make test
|
||||
|
||||
"clang":
|
||||
gcc9:
|
||||
docker:
|
||||
- image: ubuntu:18.04
|
||||
- image: conanio/gcc9
|
||||
environment:
|
||||
CXX: g++-9
|
||||
CC: gcc-9
|
||||
BUILD_FLAGS:
|
||||
CTEST_FLAGS: --output-on-failure
|
||||
|
||||
gcc10:
|
||||
docker:
|
||||
- image: conanio/gcc10
|
||||
environment:
|
||||
CXX: g++-10
|
||||
CC: gcc-10
|
||||
BUILD_FLAGS:
|
||||
CTEST_FLAGS: --output-on-failure
|
||||
|
||||
clang10:
|
||||
docker:
|
||||
- image: conanio/clang10
|
||||
environment:
|
||||
CXX: clang++-10
|
||||
CC: clang-10
|
||||
BUILD_FLAGS:
|
||||
CTEST_FLAGS: --output-on-failure
|
||||
|
||||
clang9:
|
||||
docker:
|
||||
- image: conanio/clang9
|
||||
environment:
|
||||
CXX: clang++-9
|
||||
CC: clang-9
|
||||
BUILD_FLAGS:
|
||||
CTEST_FLAGS: --output-on-failure
|
||||
|
||||
clang6:
|
||||
docker:
|
||||
- image: conanio/clang60
|
||||
environment:
|
||||
CXX: clang++-6.0
|
||||
CC: clang-6.0
|
||||
BUILD_FLAGS:
|
||||
CTEST_FLAGS: --output-on-failure
|
||||
|
||||
# Reusable test commands (and initializer for clang 6)
|
||||
commands:
|
||||
dependency_restore:
|
||||
steps:
|
||||
- restore_cache:
|
||||
keys:
|
||||
- cmake-cache-{{ checksum "dependencies/CMakeLists.txt" }}
|
||||
|
||||
dependency_cache:
|
||||
steps:
|
||||
- save_cache:
|
||||
key: cmake-cache-{{ checksum "dependencies/CMakeLists.txt" }}
|
||||
paths:
|
||||
- dependencies/.cache
|
||||
|
||||
install_cmake:
|
||||
steps:
|
||||
- run: apt-get update -qq
|
||||
- run: apt-get install -y cmake
|
||||
|
||||
cmake_prep:
|
||||
steps:
|
||||
- checkout
|
||||
- run: mkdir -p build
|
||||
|
||||
- run: apt-get update -qq
|
||||
- run: >
|
||||
apt-get install -y
|
||||
build-essential
|
||||
cmake
|
||||
clang-6.0
|
||||
cmake_build_cache:
|
||||
steps:
|
||||
- cmake_prep
|
||||
- dependency_restore
|
||||
- run: cmake $CMAKE_FLAGS -DCMAKE_INSTALL_PREFIX:PATH=destination -B build .
|
||||
- dependency_cache # dependencies are produced in the configure step
|
||||
|
||||
- run:
|
||||
name: Building (clang)
|
||||
command: make
|
||||
cmake_build:
|
||||
steps:
|
||||
- cmake_build_cache
|
||||
- run: cmake --build build
|
||||
|
||||
- run:
|
||||
name: Running tests (clang)
|
||||
command: make quiettest
|
||||
cmake_test:
|
||||
steps:
|
||||
- cmake_build
|
||||
- run: |
|
||||
cd build &&
|
||||
tools/json2json -h &&
|
||||
ctest $CTEST_FLAGS -L acceptance &&
|
||||
ctest $CTEST_FLAGS -LE acceptance -E checkperf
|
||||
|
||||
- run:
|
||||
name: Building (clang, cmake)
|
||||
command: |
|
||||
mkdir build
|
||||
cd build
|
||||
cmake ..
|
||||
make
|
||||
cmake_test_all:
|
||||
steps:
|
||||
- cmake_build
|
||||
- run: |
|
||||
cd build &&
|
||||
tools/json2json -h &&
|
||||
ctest $CTEST_FLAGS -DSIMDJSON_IMPLEMENTATION="haswell;westmere;fallback" -L acceptance -LE per_implementation &&
|
||||
SIMDJSON_FORCE_IMPLEMENTATION=haswell ctest $CTEST_FLAGS -L per_implementation -E checkperf &&
|
||||
SIMDJSON_FORCE_IMPLEMENTATION=westmere ctest $CTEST_FLAGS -L per_implementation -E checkperf &&
|
||||
SIMDJSON_FORCE_IMPLEMENTATION=fallback ctest $CTEST_FLAGS -L per_implementation -E checkperf &&
|
||||
ctest $CTEST_FLAGS -LE "acceptance|per_implementation" # Everything we haven't run yet, run now.
|
||||
|
||||
- run:
|
||||
name: Running tests (clang, cmake)
|
||||
command: |
|
||||
cd build
|
||||
make test
|
||||
|
||||
cmake_perftest:
|
||||
steps:
|
||||
- cmake_build_cache
|
||||
- run: |
|
||||
cmake --build build --target checkperf &&
|
||||
cd build &&
|
||||
ctest --output-on-failure -R checkperf
|
||||
|
||||
# we not only want cmake to build and run tests, but we want also a successful installation from which we can build, link and run programs
|
||||
cmake_install_test: # this version builds, install, test and then verify from the installation
|
||||
steps:
|
||||
- run: cd build && make install
|
||||
- run: echo -e '#include <simdjson.h>\nint main(int argc,char**argv) {simdjson::dom::parser parser;simdjson::dom::element tweets = parser.load(argv[1]); }' > tmp.cpp && c++ -Ibuild/destination/include -Lbuild/destination/lib -std=c++17 -Wl,-rpath,build/destination/lib -o linkandrun tmp.cpp -lsimdjson && ./linkandrun jsonexamples/twitter.json
|
||||
|
||||
cmake_installed_test_cxx20: # assuming that it was installed, this tries to build using C++20
|
||||
steps:
|
||||
- run: echo -e '#include <simdjson.h>\nint main(int argc,char**argv) {simdjson::dom::parser parser;simdjson::dom::element tweets = parser.load(argv[1]); }' > tmp.cpp && c++ -Ibuild/destination/include -Lbuild/destination/lib -std=c++20 -Wl,-rpath,build/destination/lib -o linkandrun tmp.cpp -lsimdjson && ./linkandrun jsonexamples/twitter.json
|
||||
|
||||
jobs:
|
||||
|
||||
# static
|
||||
justlib-gcc10:
|
||||
description: Build just the library, install it and do a basic test
|
||||
executor: gcc10
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_JUST_LIBRARY=ON }
|
||||
steps: [ cmake_build, cmake_install_test, cmake_installed_test_cxx20 ]
|
||||
gcc10-perftest:
|
||||
description: Build and run performance tests on GCC 10 and AVX 2 with a cmake static build, this test performance regression
|
||||
executor: gcc10
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_GOOGLE_BENCHMARKS=OFF -DSIMDJSON_BUILD_STATIC=ON }
|
||||
steps: [ cmake_perftest ]
|
||||
gcc10:
|
||||
description: Build and run tests on GCC 10 and AVX 2 with a cmake static build
|
||||
executor: gcc10
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_GOOGLE_BENCHMARKS=ON -DSIMDJSON_BUILD_STATIC=ON }
|
||||
steps: [ cmake_test, cmake_install_test, cmake_installed_test_cxx20 ]
|
||||
clang6:
|
||||
description: Build and run tests on clang 6 and AVX 2 with a cmake static build
|
||||
executor: clang6
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_GOOGLE_BENCHMARKS=ON -DSIMDJSON_BUILD_STATIC=ON }
|
||||
steps: [ cmake_test, cmake_install_test ]
|
||||
clang10:
|
||||
description: Build and run tests on clang 10 and AVX 2 with a cmake static build
|
||||
executor: clang10
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_GOOGLE_BENCHMARKS=ON -DSIMDJSON_BUILD_STATIC=ON }
|
||||
steps: [ cmake_test, cmake_install_test, cmake_installed_test_cxx20 ]
|
||||
# libcpp
|
||||
libcpp-clang10:
|
||||
description: Build and run tests on clang 10 and AVX 2 with a cmake static build and libc++
|
||||
executor: clang10
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_USE_LIBCPP=ON -DSIMDJSON_BUILD_STATIC=ON }
|
||||
steps: [ cmake_test, cmake_install_test, cmake_installed_test_cxx20 ]
|
||||
# sanitize
|
||||
sanitize-gcc10:
|
||||
description: Build and run tests on GCC 10 and AVX 2 with a cmake sanitize build
|
||||
executor: gcc10
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_BUILD_STATIC=OFF -DSIMDJSON_SANITIZE=ON, BUILD_FLAGS: "", CTEST_FLAGS: --output-on-failure -E checkperf }
|
||||
steps: [ cmake_test ]
|
||||
sanitize-clang10:
|
||||
description: Build and run tests on clang 10 and AVX 2 with a cmake sanitize build
|
||||
executor: clang10
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_BUILD_STATIC=OFF -DSIMDJSON_SANITIZE=ON, CTEST_FLAGS: --output-on-failure -E checkperf }
|
||||
steps: [ cmake_test ]
|
||||
threadsanitize-gcc10:
|
||||
description: Build and run tests on GCC 10 and AVX 2 with a cmake sanitize build
|
||||
executor: gcc10
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_BUILD_STATIC=OFF -DSIMDJSON_SANITIZE_THREADS=ON, BUILD_FLAGS: "", CTEST_FLAGS: --output-on-failure -E checkperf }
|
||||
steps: [ cmake_test ]
|
||||
threadsanitize-clang10:
|
||||
description: Build and run tests on clang 10 and AVX 2 with a cmake sanitize build
|
||||
executor: clang10
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_BUILD_STATIC=OFF -DSIMDJSON_SANITIZE_THREADS=ON, CTEST_FLAGS: --output-on-failure -E checkperf }
|
||||
steps: [ cmake_test ]
|
||||
# dynamic
|
||||
dynamic-gcc10:
|
||||
description: Build and run tests on GCC 10 and AVX 2 with a cmake dynamic build
|
||||
executor: gcc10
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_BUILD_STATIC=OFF }
|
||||
steps: [ cmake_test, cmake_install_test ]
|
||||
dynamic-clang10:
|
||||
description: Build and run tests on clang 10 and AVX 2 with a cmake dynamic build
|
||||
executor: clang10
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_BUILD_STATIC=OFF }
|
||||
steps: [ cmake_test, cmake_install_test ]
|
||||
|
||||
# unthreaded
|
||||
unthreaded-gcc10:
|
||||
description: Build and run tests on GCC 10 and AVX 2 *without* threads
|
||||
executor: gcc10
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_ENABLE_THREADS=OFF }
|
||||
steps: [ cmake_test, cmake_install_test ]
|
||||
unthreaded-clang10:
|
||||
description: Build and run tests on Clang 10 and AVX 2 *without* threads
|
||||
executor: clang10
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_ENABLE_THREADS=OFF }
|
||||
steps: [ cmake_test, cmake_install_test ]
|
||||
|
||||
# noexcept
|
||||
noexcept-gcc10:
|
||||
description: Build and run tests on GCC 10 and AVX 2 with exceptions off
|
||||
executor: gcc10
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_EXCEPTIONS=OFF }
|
||||
steps: [ cmake_test, cmake_install_test ]
|
||||
noexcept-clang10:
|
||||
description: Build and run tests on Clang 10 and AVX 2 with exceptions off
|
||||
executor: clang10
|
||||
environment: { CMAKE_FLAGS: -DSIMDJSON_EXCEPTIONS=OFF }
|
||||
steps: [ cmake_test, cmake_install_test ]
|
||||
|
||||
#
|
||||
# Misc.
|
||||
#
|
||||
|
||||
# make (test and checkperf)
|
||||
arch-haswell-gcc10:
|
||||
description: Build, run tests and check performance on GCC 10 with -march=haswell
|
||||
executor: gcc10
|
||||
environment: { CXXFLAGS: -march=haswell }
|
||||
steps: [ cmake_test ]
|
||||
arch-nehalem-gcc10:
|
||||
description: Build, run tests and check performance on GCC 10 with -march=nehalem
|
||||
executor: gcc10
|
||||
environment: { CXXFLAGS: -march=nehalem }
|
||||
steps: [ cmake_test ]
|
||||
sanitize-haswell-gcc10:
|
||||
description: Build and run tests on GCC 10 and AVX 2 with a cmake sanitize build
|
||||
executor: gcc10
|
||||
environment: { CXXFLAGS: -march=haswell, CMAKE_FLAGS: -DSIMDJSON_BUILD_STATIC=OFF -DSIMDJSON_SANITIZE=ON, BUILD_FLAGS: "", CTEST_FLAGS: --output-on-failure -E checkperf }
|
||||
steps: [ cmake_test ]
|
||||
sanitize-haswell-clang10:
|
||||
description: Build and run tests on clang 10 and AVX 2 with a cmake sanitize build
|
||||
executor: clang10
|
||||
environment: { CXXFLAGS: -march=haswell, CMAKE_FLAGS: -DSIMDJSON_BUILD_STATIC=OFF -DSIMDJSON_SANITIZE=ON, CTEST_FLAGS: --output-on-failure -E checkperf }
|
||||
steps: [ cmake_test ]
|
||||
|
||||
workflows:
|
||||
version: 2
|
||||
version: 2.1
|
||||
build_and_test:
|
||||
jobs:
|
||||
- "clang"
|
||||
- "gcc"
|
||||
# full multi-implementation tests
|
||||
#- gcc7 tested on GitHub actions
|
||||
- gcc10 # do not delete this as it tests our performance
|
||||
- clang6
|
||||
#- clang10 # this gets tested a lot below
|
||||
|
||||
# libc++
|
||||
- libcpp-clang10
|
||||
|
||||
# full single-implementation tests
|
||||
- sanitize-gcc10
|
||||
- sanitize-clang10
|
||||
- threadsanitize-gcc10
|
||||
- threadsanitize-clang10
|
||||
- dynamic-gcc10
|
||||
- dynamic-clang10
|
||||
- unthreaded-gcc10
|
||||
- unthreaded-clang10
|
||||
|
||||
# no exceptions
|
||||
- noexcept-gcc10
|
||||
- noexcept-clang10
|
||||
|
||||
# quicker make single-implementation tests
|
||||
- arch-haswell-gcc10
|
||||
- arch-nehalem-gcc10
|
||||
|
||||
|
||||
# sanitized single-implementation tests
|
||||
- sanitize-haswell-gcc10
|
||||
- sanitize-haswell-clang10
|
||||
|
||||
# testing "just the library"
|
||||
- justlib-gcc10
|
||||
|
||||
# TODO add windows: https://circleci.com/docs/2.0/configuration-reference/#windows
|
||||
|
||||
+26
@@ -0,0 +1,26 @@
|
||||
task:
|
||||
timeout_in: 120m
|
||||
freebsd_instance:
|
||||
matrix:
|
||||
- image_family: freebsd-13-0-snap
|
||||
|
||||
env:
|
||||
ASSUME_ALWAYS_YES: YES
|
||||
simdjson_DEPENDENCY_CACHE_DIR: $HOME/.dep_cache
|
||||
dep_cache:
|
||||
folder: $HOME/.dep_cache
|
||||
reupload_on_changes: false
|
||||
fingerprint_script: cat dependencies/CMakeLists.txt
|
||||
setup_script:
|
||||
- pkg update -f
|
||||
- pkg install bash
|
||||
- pkg install cmake
|
||||
- pkg install git
|
||||
build_script:
|
||||
- mkdir build
|
||||
- cd build
|
||||
- cmake -DSIMDJSON_BASH=OFF -DSIMDJSON_GIT=OFF ..
|
||||
- make
|
||||
test_script:
|
||||
- cd build
|
||||
- ctest --output-on-failure -E checkperf
|
||||
@@ -0,0 +1 @@
|
||||
BasedOnStyle: LLVM
|
||||
@@ -0,0 +1,15 @@
|
||||
*
|
||||
!.git
|
||||
!Makefile
|
||||
!amalgamate.py
|
||||
!benchmark
|
||||
!dependencies
|
||||
!include
|
||||
!jsonchecker
|
||||
!jsonexamples
|
||||
!scripts
|
||||
!singleheader
|
||||
!src
|
||||
!style
|
||||
!tests
|
||||
!tools
|
||||
+459
-6
@@ -1,9 +1,462 @@
|
||||
kind: pipeline
|
||||
name: default
|
||||
|
||||
name: i386-gcc # we do not support 32-bit systems, but we run tests
|
||||
platform: { os: linux, arch: amd64 }
|
||||
steps:
|
||||
- name: test
|
||||
image: gcc:8
|
||||
- name: Build and Test
|
||||
image: i386/ubuntu
|
||||
environment:
|
||||
CC: gcc
|
||||
CXX: g++
|
||||
BUILD_FLAGS: -- -j
|
||||
CMAKE_FLAGS: -DSIMDJSON_BUILD_STATIC=ON
|
||||
CTEST_FLAGS: -j4 --output-on-failure -E checkperf
|
||||
commands:
|
||||
- make -j2
|
||||
- make quiettest -j2
|
||||
- scripts/addcmakeppa.sh "$(env -i sh -c '. /etc/os-release; echo $VERSION_CODENAME')"
|
||||
- apt-get install -y g++ cmake gcc git
|
||||
- mkdir build
|
||||
- cd build
|
||||
- cmake $CMAKE_FLAGS ..
|
||||
- cmake --build . $BUILD_FLAGS
|
||||
- ctest $CTEST_FLAGS
|
||||
---
|
||||
kind: pipeline
|
||||
name: i386-clang # we do not support 32-bit systems, but we run tests
|
||||
platform: { os: linux, arch: amd64 }
|
||||
steps:
|
||||
- name: Build and Test
|
||||
image: i386/ubuntu
|
||||
environment:
|
||||
CC: clang-6.0
|
||||
CXX: clang++-6.0
|
||||
BUILD_FLAGS: -- -j
|
||||
CMAKE_FLAGS: -DSIMDJSON_BUILD_STATIC=ON
|
||||
CTEST_FLAGS: -j4 --output-on-failure -E checkperf
|
||||
commands:
|
||||
- scripts/addcmakeppa.sh "$(env -i sh -c '. /etc/os-release; echo $VERSION_CODENAME')"
|
||||
- apt-get install -y clang++-6.0 cmake git
|
||||
- mkdir build
|
||||
- cd build
|
||||
- cmake $CMAKE_FLAGS ..
|
||||
- cmake --build . $BUILD_FLAGS
|
||||
- ctest $CTEST_FLAGS
|
||||
---
|
||||
kind: pipeline
|
||||
name: gcc9
|
||||
platform: { os: linux, arch: amd64 }
|
||||
steps:
|
||||
- name: Build and Test
|
||||
image: gcc:9
|
||||
environment:
|
||||
CC: gcc
|
||||
CXX: g++
|
||||
BUILD_FLAGS: -- -j
|
||||
CMAKE_FLAGS: -DSIMDJSON_BUILD_STATIC=ON -DSIMDJSON_IMPLEMENTATION=haswell;westmere;fallback
|
||||
CTEST_FLAGS: -j4 --output-on-failure -E checkperf
|
||||
commands:
|
||||
- echo "deb http://deb.debian.org/debian buster-backports main" >> /etc/apt/sources.list
|
||||
- apt-get update -qq
|
||||
- apt-get -t buster-backports install -y cmake
|
||||
- mkdir build
|
||||
- cd build
|
||||
- cmake $CMAKE_FLAGS ..
|
||||
- cmake --build . $BUILD_FLAGS
|
||||
- ctest $CTEST_FLAGS -L acceptance -LE per_implementation
|
||||
- SIMDJSON_FORCE_IMPLEMENTATION=haswell ctest $CTEST_FLAGS -L per_implementation
|
||||
- SIMDJSON_FORCE_IMPLEMENTATION=westmere ctest $CTEST_FLAGS -L per_implementation
|
||||
- SIMDJSON_FORCE_IMPLEMENTATION=fallback ctest $CTEST_FLAGS -L per_implementation
|
||||
- ctest $CTEST_FLAGS -LE "acceptance|per_implementation" # Everything we haven't run yet, run now.
|
||||
---
|
||||
kind: pipeline
|
||||
name: clang6
|
||||
platform: { os: linux, arch: amd64 }
|
||||
steps:
|
||||
- name: Build and Test
|
||||
image: conanio/clang60
|
||||
user: root
|
||||
environment:
|
||||
CC: clang-6.0
|
||||
CXX: clang++-6.0
|
||||
BUILD_FLAGS: -- -j
|
||||
CMAKE_FLAGS: -DSIMDJSON_BUILD_STATIC=ON -DSIMDJSON_IMPLEMENTATION=haswell;westmere;fallback
|
||||
CTEST_FLAGS: -j4 --output-on-failure -E checkperf
|
||||
commands:
|
||||
- mkdir build
|
||||
- cd build
|
||||
- cmake $CMAKE_FLAGS ..
|
||||
- cmake --build . $BUILD_FLAGS
|
||||
- ctest $CTEST_FLAGS -L acceptance -LE per_implementation
|
||||
- SIMDJSON_FORCE_IMPLEMENTATION=haswell ctest $CTEST_FLAGS -L per_implementation
|
||||
- SIMDJSON_FORCE_IMPLEMENTATION=westmere ctest $CTEST_FLAGS -L per_implementation
|
||||
- SIMDJSON_FORCE_IMPLEMENTATION=fallback ctest $CTEST_FLAGS -L per_implementation
|
||||
- ctest $CTEST_FLAGS -LE "acceptance|per_implementation" # Everything we haven't run yet, run now.
|
||||
---
|
||||
kind: pipeline
|
||||
name: dynamic-gcc9
|
||||
platform: { os: linux, arch: amd64 }
|
||||
steps:
|
||||
- name: Build and Test
|
||||
image: gcc:9
|
||||
environment:
|
||||
CC: gcc
|
||||
CXX: g++
|
||||
BUILD_FLAGS: -- -j
|
||||
CMAKE_FLAGS: -DSIMDJSON_BUILD_STATIC=OFF
|
||||
CTEST_FLAGS: -j4 --output-on-failure -E checkperf
|
||||
commands:
|
||||
- echo "deb http://deb.debian.org/debian buster-backports main" >> /etc/apt/sources.list
|
||||
- apt-get update -qq
|
||||
- apt-get -t buster-backports install -y cmake
|
||||
- mkdir build
|
||||
- cd build
|
||||
- cmake $CMAKE_FLAGS ..
|
||||
- cmake --build . $BUILD_FLAGS
|
||||
- ctest $CTEST_FLAGS
|
||||
---
|
||||
kind: pipeline
|
||||
name: dynamic-clang9
|
||||
platform: { os: linux, arch: amd64 }
|
||||
steps:
|
||||
- name: Build and Test
|
||||
image: conanio/clang9
|
||||
user: root
|
||||
environment:
|
||||
CC: clang-9
|
||||
CXX: clang++-9
|
||||
CMAKE_FLAGS: -DSIMDJSON_BUILD_STATIC=OFF
|
||||
BUILD_FLAGS: -- -j
|
||||
CTEST_FLAGS: -j4 --output-on-failure -E checkperf
|
||||
commands:
|
||||
- mkdir build
|
||||
- cd build
|
||||
- cmake $CMAKE_FLAGS ..
|
||||
- cmake --build . $BUILD_FLAGS
|
||||
- ctest $CTEST_FLAGS
|
||||
---
|
||||
kind: pipeline
|
||||
name: sanitize-gcc9
|
||||
platform: { os: linux, arch: amd64 }
|
||||
steps:
|
||||
- name: Build and Test
|
||||
image: gcc:9
|
||||
environment:
|
||||
CC: gcc
|
||||
CXX: g++
|
||||
BUILD_FLAGS: -- -j
|
||||
CMAKE_FLAGS: -DSIMDJSON_BUILD_STATIC=ON -DSIMDJSON_IMPLEMENTATION=haswell;westmere;fallback
|
||||
CTEST_FLAGS: -j4 --output-on-failure -E checkperf
|
||||
commands:
|
||||
- echo "deb http://deb.debian.org/debian buster-backports main" >> /etc/apt/sources.list
|
||||
- apt-get update -qq
|
||||
- apt-get -t buster-backports install -y cmake
|
||||
- mkdir build
|
||||
- cd build
|
||||
- cmake $CMAKE_FLAGS ..
|
||||
- cmake --build . $BUILD_FLAGS
|
||||
- ASAN_OPTIONS="detect_leaks=0" ctest $CTEST_FLAGS -L acceptance -LE per_implementation
|
||||
- SIMDJSON_FORCE_IMPLEMENTATION=haswell ASAN_OPTIONS="detect_leaks=0" ctest $CTEST_FLAGS -L per_implementation
|
||||
- SIMDJSON_FORCE_IMPLEMENTATION=westmere ASAN_OPTIONS="detect_leaks=0" ctest $CTEST_FLAGS -L per_implementation
|
||||
- SIMDJSON_FORCE_IMPLEMENTATION=fallback ASAN_OPTIONS="detect_leaks=0" ctest $CTEST_FLAGS -L per_implementation
|
||||
- ASAN_OPTIONS="detect_leaks=0" ctest $CTEST_FLAGS -LE "acceptance|per_implementation" # Everything we haven't run yet, run now.
|
||||
---
|
||||
kind: pipeline
|
||||
name: sanitize-clang9
|
||||
platform: { os: linux, arch: amd64 }
|
||||
steps:
|
||||
- name: Build and Test
|
||||
image: conanio/clang9
|
||||
user: root
|
||||
environment:
|
||||
CC: clang-9
|
||||
CXX: clang++-9
|
||||
CMAKE_FLAGS: -DSIMDJSON_SANITIZE=ON -DSIMDJSON_IMPLEMENTATION=haswell;westmere;fallback
|
||||
BUILD_FLAGS: -- -j
|
||||
CTEST_FLAGS: -j4 --output-on-failure -E checkperf
|
||||
commands:
|
||||
- mkdir build
|
||||
- cd build
|
||||
- cmake $CMAKE_FLAGS ..
|
||||
- cmake --build . $BUILD_FLAGS
|
||||
- ASAN_OPTIONS="detect_leaks=0" ctest $CTEST_FLAGS -L acceptance -LE per_implementation
|
||||
- SIMDJSON_FORCE_IMPLEMENTATION=haswell ASAN_OPTIONS="detect_leaks=0" ctest $CTEST_FLAGS -L per_implementation
|
||||
- SIMDJSON_FORCE_IMPLEMENTATION=westmere ASAN_OPTIONS="detect_leaks=0" ctest $CTEST_FLAGS -L per_implementation
|
||||
- SIMDJSON_FORCE_IMPLEMENTATION=fallback ASAN_OPTIONS="detect_leaks=0" ctest $CTEST_FLAGS -L per_implementation
|
||||
- ASAN_OPTIONS="detect_leaks=0" ctest $CTEST_FLAGS -LE "acceptance|per_implementation" # Everything we haven't run yet, run now.
|
||||
---
|
||||
kind: pipeline
|
||||
name: cpp20-clang11-libcpp
|
||||
platform: { os: linux, arch: amd64 }
|
||||
steps:
|
||||
- name: Build and Test
|
||||
image: pauldreik/llvm-11
|
||||
user: root
|
||||
environment:
|
||||
CC: clang-11
|
||||
CXX: clang++-11
|
||||
CMAKE_FLAGS: -GNinja
|
||||
BUILD_FLAGS:
|
||||
CTEST_FLAGS: -j4 --output-on-failure -E checkperf
|
||||
CXXFLAGS: -std=c++20 -stdlib=libc++
|
||||
commands:
|
||||
- mkdir build
|
||||
- cd build
|
||||
- cmake $CMAKE_FLAGS ..
|
||||
- cmake --build . $BUILD_FLAGS
|
||||
- ctest $CTEST_FLAGS
|
||||
---
|
||||
kind: pipeline
|
||||
name: arm64-gcc8
|
||||
platform: { os: linux, arch: arm64 }
|
||||
steps:
|
||||
- name: Build and Test
|
||||
image: gcc:8
|
||||
environment:
|
||||
CC: gcc
|
||||
CXX: g++
|
||||
BUILD_FLAGS: -- -j
|
||||
CMAKE_FLAGS: -DSIMDJSON_BUILD_STATIC=ON -DSIMDJSON_IMPLEMENTATION=arm64;fallback
|
||||
CTEST_FLAGS: -j4 --output-on-failure -E checkperf
|
||||
commands:
|
||||
- echo "deb http://deb.debian.org/debian buster-backports main" >> /etc/apt/sources.list
|
||||
- apt-get update -qq
|
||||
- apt-get -t buster-backports install -y cmake
|
||||
- mkdir build
|
||||
- cd build
|
||||
- cmake $CMAKE_FLAGS ..
|
||||
- cmake --build . $BUILD_FLAGS
|
||||
- ctest $CTEST_FLAGS -L acceptance -LE per_implementation
|
||||
- SIMDJSON_FORCE_IMPLEMENTATION=arm64 ctest $CTEST_FLAGS -L per_implementation
|
||||
- SIMDJSON_FORCE_IMPLEMENTATION=fallback ctest $CTEST_FLAGS -L per_implementation
|
||||
- ctest $CTEST_FLAGS -LE "acceptance|per_implementation" # Everything we haven't run yet, run now.
|
||||
---
|
||||
kind: pipeline
|
||||
name: arm64-clang6
|
||||
platform: { os: linux, arch: arm64 }
|
||||
steps:
|
||||
- name: Build and Test
|
||||
image: debian:buster-backports
|
||||
environment:
|
||||
CC: clang-6.0
|
||||
CXX: clang++-6.0
|
||||
CMAKE_FLAGS: -DSIMDJSON_BUILD_STATIC=OFF
|
||||
BUILD_FLAGS: -- -j
|
||||
CTEST_FLAGS: -j4 --output-on-failure -E checkperf
|
||||
commands:
|
||||
- apt-get -qq update
|
||||
- apt-get -t buster-backports install -y cmake
|
||||
- apt-get install -y clang-6.0 git
|
||||
- mkdir build
|
||||
- cd build
|
||||
- cmake $CMAKE_FLAGS ..
|
||||
- cmake --build . $BUILD_FLAGS
|
||||
- ctest $CTEST_FLAGS
|
||||
---
|
||||
kind: pipeline
|
||||
name: arm64-dynamic-gcc8
|
||||
platform: { os: linux, arch: arm64 }
|
||||
steps:
|
||||
- name: Build and Test
|
||||
image: gcc:8
|
||||
environment:
|
||||
CC: gcc
|
||||
CXX: g++
|
||||
BUILD_FLAGS: -- -j
|
||||
CMAKE_FLAGS: -DSIMDJSON_BUILD_STATIC=OFF
|
||||
CTEST_FLAGS: -j4 --output-on-failure -E checkperf
|
||||
commands:
|
||||
- echo "deb http://deb.debian.org/debian buster-backports main" >> /etc/apt/sources.list
|
||||
- apt-get update -qq
|
||||
- apt-get -t buster-backports install -y cmake
|
||||
- mkdir build
|
||||
- cd build
|
||||
- cmake $CMAKE_FLAGS ..
|
||||
- cmake --build . $BUILD_FLAGS
|
||||
- ctest $CTEST_FLAGS
|
||||
---
|
||||
kind: pipeline
|
||||
name: arm64-dynamic-clang6
|
||||
platform: { os: linux, arch: arm64 }
|
||||
steps:
|
||||
- name: Build and Test
|
||||
image: debian:buster-backports
|
||||
environment:
|
||||
CC: clang-6.0
|
||||
CXX: clang++-6.0
|
||||
CMAKE_FLAGS: -DSIMDJSON_BUILD_STATIC=OFF
|
||||
BUILD_FLAGS: -- -j
|
||||
CTEST_FLAGS: -j4 --output-on-failure -E checkperf
|
||||
commands:
|
||||
- apt-get -qq update
|
||||
- apt-get -t buster-backports install -y cmake
|
||||
- apt-get install -y clang-6.0 git
|
||||
- mkdir build
|
||||
- cd build
|
||||
- cmake $CMAKE_FLAGS ..
|
||||
- cmake --build . $BUILD_FLAGS
|
||||
- ctest $CTEST_FLAGS
|
||||
---
|
||||
kind: pipeline
|
||||
name: arm64-sanitize-gcc8
|
||||
platform: { os: linux, arch: arm64 }
|
||||
steps:
|
||||
- name: Build and Test
|
||||
image: gcc:8
|
||||
environment:
|
||||
BUILD_FLAGS: -- -j
|
||||
CMAKE_FLAGS: -DSIMDJSON_BUILD_STATIC=ON -DSIMDJSON_IMPLEMENTATION=arm64;fallback
|
||||
CTEST_FLAGS: -j4 --output-on-failure -E checkperf
|
||||
CC: gcc
|
||||
CXX: g++
|
||||
commands:
|
||||
- echo "deb http://deb.debian.org/debian buster-backports main" >> /etc/apt/sources.list
|
||||
- apt-get update -qq
|
||||
- apt-get -t buster-backports install -y cmake
|
||||
- apt-get install -y libstdc++6
|
||||
- mkdir build
|
||||
- cd build
|
||||
- cmake $CMAKE_FLAGS ..
|
||||
- cmake --build . $BUILD_FLAGS
|
||||
- ASAN_OPTIONS="detect_leaks=0" ctest $CTEST_FLAGS -L acceptance -LE per_implementation
|
||||
- SIMDJSON_FORCE_IMPLEMENTATION=arm64 ASAN_OPTIONS="detect_leaks=0" ctest $CTEST_FLAGS -L per_implementation
|
||||
- SIMDJSON_FORCE_IMPLEMENTATION=fallback ASAN_OPTIONS="detect_leaks=0" ctest $CTEST_FLAGS -L per_implementation
|
||||
- ASAN_OPTIONS="detect_leaks=0" ctest $CTEST_FLAGS -LE "acceptance|per_implementation" # Everything we haven't run yet, run now.
|
||||
---
|
||||
kind: pipeline
|
||||
name: arm64-sanitize-clang6
|
||||
platform: { os: linux, arch: arm64 }
|
||||
steps:
|
||||
- name: Build and Test
|
||||
image: debian:buster-backports
|
||||
environment:
|
||||
CC: clang-6.0
|
||||
CXX: clang++-6.0
|
||||
CMAKE_FLAGS: -DSIMDJSON_SANITIZE=ON -DSIMDJSON_IMPLEMENTATION=arm64;fallback
|
||||
BUILD_FLAGS: -- -j
|
||||
CTEST_FLAGS: -j4 --output-on-failure -E checkperf
|
||||
commands:
|
||||
- apt-get -qq update
|
||||
- apt-get -t buster-backports install -y cmake
|
||||
- apt-get install -y clang-6.0 git
|
||||
- mkdir build
|
||||
- cd build
|
||||
- cmake $CMAKE_FLAGS ..
|
||||
- cmake --build . $BUILD_FLAGS
|
||||
- ASAN_OPTIONS="detect_leaks=0" ctest $CTEST_FLAGS -L acceptance -LE per_implementation
|
||||
- SIMDJSON_FORCE_IMPLEMENTATION=arm64 ASAN_OPTIONS="detect_leaks=0" ctest $CTEST_FLAGS -L per_implementation
|
||||
- SIMDJSON_FORCE_IMPLEMENTATION=fallback ASAN_OPTIONS="detect_leaks=0" ctest $CTEST_FLAGS -L per_implementation
|
||||
- ASAN_OPTIONS="detect_leaks=0" ctest $CTEST_FLAGS -LE "acceptance|per_implementation" # Everything we haven't run yet, run now.
|
||||
---
|
||||
kind: pipeline
|
||||
name: ninja-clang9
|
||||
platform: { os: linux, arch: amd64 }
|
||||
steps:
|
||||
- name: Build and Test
|
||||
image: conanio/clang9
|
||||
user: root
|
||||
environment:
|
||||
CC: clang-9
|
||||
CXX: clang++-9
|
||||
BUILD_FLAGS: -- -j 4
|
||||
CMAKE_FLAGS: -GNinja -DSIMDJSON_BUILD_STATIC=ON
|
||||
CTEST_FLAGS: -j4 --output-on-failure -E checkperf
|
||||
CXXFLAGS: -stdlib=libc++
|
||||
commands:
|
||||
- mkdir build
|
||||
- cd build
|
||||
- cmake $CMAKE_FLAGS ..
|
||||
- cmake --build . $BUILD_FLAGS
|
||||
- ctest $CTEST_FLAGS
|
||||
---
|
||||
kind: pipeline
|
||||
name: libcpp-clang9
|
||||
platform: { os: linux, arch: amd64 }
|
||||
steps:
|
||||
- name: Build and Test
|
||||
image: conanio/clang9
|
||||
user: root
|
||||
environment:
|
||||
CC: clang-9
|
||||
CXX: clang++-9
|
||||
BUILD_FLAGS: -- -j
|
||||
CMAKE_FLAGS: -DSIMDJSON_BUILD_STATIC=ON
|
||||
CTEST_FLAGS: -j4 --output-on-failure -E checkperf
|
||||
CXXFLAGS: -stdlib=libc++
|
||||
commands:
|
||||
- mkdir build
|
||||
- cd build
|
||||
- cmake $CMAKE_FLAGS ..
|
||||
- cmake --build . $BUILD_FLAGS
|
||||
- ctest $CTEST_FLAGS
|
||||
---
|
||||
kind: pipeline
|
||||
name: libcpp-clang7
|
||||
platform: { os: linux, arch: amd64 }
|
||||
steps:
|
||||
- name: Build and Test
|
||||
image: conanio/clang7
|
||||
user: root
|
||||
environment:
|
||||
CC: clang-7
|
||||
CXX: clang++-7
|
||||
BUILD_FLAGS: -- -j
|
||||
CMAKE_FLAGS: -DSIMDJSON_BUILD_STATIC=ON
|
||||
CTEST_FLAGS: -j4 --output-on-failure -E checkperf
|
||||
CXXFLAGS: -stdlib=libc++
|
||||
commands:
|
||||
- mkdir build
|
||||
- cd build
|
||||
- cmake $CMAKE_FLAGS ..
|
||||
- cmake --build . $BUILD_FLAGS
|
||||
- ctest $CTEST_FLAGS
|
||||
---
|
||||
kind: pipeline
|
||||
name: noexceptions-gcc9
|
||||
platform: { os: linux, arch: amd64 }
|
||||
steps:
|
||||
- name: Build and Test
|
||||
image: gcc:9
|
||||
environment:
|
||||
CC: gcc
|
||||
CXX: g++
|
||||
BUILD_FLAGS: -- -j
|
||||
CMAKE_FLAGS: -DSIMDJSON_EXCEPTIONS=OFF
|
||||
CTEST_FLAGS: -j4 --output-on-failure -E checkperf
|
||||
commands:
|
||||
- echo "deb http://deb.debian.org/debian buster-backports main" >> /etc/apt/sources.list
|
||||
- apt-get update -qq
|
||||
- apt-get -t buster-backports install -y cmake
|
||||
- mkdir build
|
||||
- cd build
|
||||
- cmake $CMAKE_FLAGS ..
|
||||
- cmake --build . $BUILD_FLAGS
|
||||
- ctest $CTEST_FLAGS
|
||||
---
|
||||
kind: pipeline
|
||||
name: arm64-fuzz
|
||||
platform: { os: linux, arch: arm64 }
|
||||
steps:
|
||||
- name: Build and run fuzzers shortly
|
||||
image: ubuntu:20.04
|
||||
environment:
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
DEBIAN_FRONTEND: noninteractive
|
||||
ASAN_OPTIONS: detect_leaks=0
|
||||
commands:
|
||||
- apt-get -qq update
|
||||
- apt-get install -q -y clang cmake git wget zip ninja-build
|
||||
- wget --quiet https://dl.bintray.com/pauldreik/simdjson-fuzz-corpus/corpus/corpus.tar
|
||||
- tar xf corpus.tar && rm corpus.tar
|
||||
- fuzz/build_like_ossfuzz.sh
|
||||
- mkdir -p common_out
|
||||
- for fuzzer in build/fuzz/fuzz_* ; do echo $fuzzer;$fuzzer common_out out/* -max_total_time=40; done
|
||||
---
|
||||
kind: pipeline
|
||||
name: stylecheck
|
||||
platform: { os: linux, arch: amd64 }
|
||||
steps:
|
||||
- name: Build and Test
|
||||
image: ubuntu:18.04
|
||||
commands:
|
||||
- apt-get update -y
|
||||
- apt-get install -y python clang-format
|
||||
- ./style/run-clang-format.py -r include/ benchmark/ src/ tests/
|
||||
|
||||
+113
-15
@@ -1,19 +1,117 @@
|
||||
# Set the default behavior, in case people don't have core.autocrlf set.
|
||||
# Uncomment next line to adjust line endings
|
||||
#* text=auto eol=lf
|
||||
* text=auto
|
||||
|
||||
# Explicitly declare text files you want to always be normalized and converted
|
||||
# to native line endings on checkout.
|
||||
*.c text
|
||||
*.cpp text
|
||||
*.h text
|
||||
*.java text
|
||||
*.xml text
|
||||
|
||||
|
||||
# Denote all files that are truly binary and should not be modified.
|
||||
*.png binary
|
||||
*.jpg binary
|
||||
*.svg binary
|
||||
*.json binary
|
||||
# we don't want json files to be modified for this project
|
||||
*.json binary
|
||||
|
||||
|
||||
# Common settings that generally should always be used with your language specific settings
|
||||
|
||||
#
|
||||
# From generator:
|
||||
# https://www.davidlaing.com/2012/09/19/customise-your-gitattributes-to-become-a-git-ninja/
|
||||
#
|
||||
|
||||
# Documents
|
||||
*.bibtex text diff=bibtex
|
||||
*.doc diff=astextplain
|
||||
*.DOC diff=astextplain
|
||||
*.docx diff=astextplain
|
||||
*.DOCX diff=astextplain
|
||||
*.dot diff=astextplain
|
||||
*.DOT diff=astextplain
|
||||
*.pdf diff=astextplain
|
||||
*.PDF diff=astextplain
|
||||
*.rtf diff=astextplain
|
||||
*.RTF diff=astextplain
|
||||
*.md text
|
||||
*.tex text diff=tex
|
||||
*.adoc text
|
||||
*.textile text
|
||||
*.mustache text
|
||||
*.csv text
|
||||
*.tab text
|
||||
*.tsv text
|
||||
*.txt text
|
||||
*.sql text
|
||||
|
||||
# Graphics
|
||||
*.png binary
|
||||
*.jpg binary
|
||||
*.jpeg binary
|
||||
*.gif binary
|
||||
*.tif binary
|
||||
*.tiff binary
|
||||
*.ico binary
|
||||
# SVG treated as an asset (binary) by default.
|
||||
*.svg text
|
||||
# If you want to treat it as binary,
|
||||
# use the following line instead.
|
||||
# *.svg binary
|
||||
*.eps binary
|
||||
|
||||
# Scripts
|
||||
*.bash text eol=lf
|
||||
*.sh text eol=lf
|
||||
# These are explicitly windows files and should use crlf
|
||||
*.bat text eol=crlf
|
||||
*.cmd text eol=crlf
|
||||
*.ps1 text eol=crlf
|
||||
|
||||
# Serialisation
|
||||
#*.json text
|
||||
*.toml text
|
||||
*.xml text
|
||||
*.yaml text
|
||||
*.yml text
|
||||
|
||||
# Archives
|
||||
*.7z binary
|
||||
*.gz binary
|
||||
*.tar binary
|
||||
*.zip binary
|
||||
|
||||
#
|
||||
# Exclude files from exporting
|
||||
#
|
||||
|
||||
.gitattributes export-ignore
|
||||
.gitignore export-ignore
|
||||
|
||||
# Sources
|
||||
*.c text eol=lf diff=c
|
||||
*.cc text eol=lf diff=cpp
|
||||
*.cxx text eol=lf diff=cpp
|
||||
*.cpp text eol=lf diff=cpp
|
||||
*.c++ text eol=lf diff=cpp
|
||||
*.hpp text eol=lf diff=cpp
|
||||
*.h text eol=lf diff=c
|
||||
*.h++ text eol=lf diff=cpp
|
||||
*.hh text eol=lf diff=cpp
|
||||
|
||||
# Compiled Object files
|
||||
*.slo binary
|
||||
*.lo binary
|
||||
*.o binary
|
||||
*.obj binary
|
||||
|
||||
# Precompiled Headers
|
||||
*.gch binary
|
||||
*.pch binary
|
||||
|
||||
# Compiled Dynamic libraries
|
||||
*.so binary
|
||||
*.dylib binary
|
||||
*.dll binary
|
||||
|
||||
# Compiled Static libraries
|
||||
*.lai binary
|
||||
*.la binary
|
||||
*.a binary
|
||||
*.lib binary
|
||||
|
||||
# Executables
|
||||
*.exe binary
|
||||
*.out binary
|
||||
*.app binary
|
||||
|
||||
@@ -0,0 +1,41 @@
|
||||
---
|
||||
name: Bug report
|
||||
about: Create a report to help us improve
|
||||
title: ''
|
||||
labels: bug (unverified)
|
||||
assignees: ''
|
||||
|
||||
---
|
||||
|
||||
Before submitting an issue, please ensure that you have read the documentation:
|
||||
|
||||
* Basics is an overview of how to use simdjson and its APIs: https://github.com/simdjson/simdjson/blob/master/doc/basics.md
|
||||
* Performance shows some more advanced scenarios and how to tune for them: https://github.com/simdjson/simdjson/blob/master/doc/performance.md
|
||||
* Contributing: https://github.com/simdjson/simdjson/blob/master/CONTRIBUTING.md
|
||||
* We follow the [JSON specification as described by RFC 8259](https://www.rfc-editor.org/rfc/rfc8259.txt) (T. Bray, 2017).
|
||||
|
||||
|
||||
**Describe the bug**
|
||||
A clear and concise description of what the bug is.
|
||||
|
||||
Note that a compiler warning is not a bug.
|
||||
|
||||
**To Reproduce**
|
||||
Steps to reproduce the behaviour: provide a code sample if possible.
|
||||
|
||||
If we cannot reproduce the issue, then we cannot address it.
|
||||
|
||||
Note that a stack trace from your own program is not enough.
|
||||
|
||||
**Configuration (please complete the following information if relevant):**
|
||||
- OS: [e.g. Ubuntu 16.04.6 LTS]
|
||||
- Compiler [e.g. Apple clang version 11.0.3 (clang-1103.0.32.59) x86_64-apple-darwin19.4.0]
|
||||
- Version [e.g. 22]
|
||||
|
||||
We support up-to-date 64-bit ARM and x64 FreeBSD, macOS, Windows and Linux systems. Please ensure that your configuration is supported before labelling the issue as a bug. In particular, we do not support legacy 32-bit systems.
|
||||
|
||||
**Indicate whether you are willing or able to provide a bug fix as a pull request**
|
||||
|
||||
If you plan to contribute to simdjson, please read our
|
||||
* CONTRIBUTING guide: https://github.com/simdjson/simdjson/blob/master/CONTRIBUTING.md and our
|
||||
* HACKING guide: https://github.com/simdjson/simdjson/blob/master/HACKING.md
|
||||
@@ -0,0 +1,37 @@
|
||||
---
|
||||
name: Feature request
|
||||
about: Suggest an idea for this project
|
||||
title: ''
|
||||
labels: ''
|
||||
assignees: ''
|
||||
|
||||
---
|
||||
|
||||
Before submitting an issue, please ensure that you have read the documentation:
|
||||
|
||||
* Basics is an overview of how to use simdjson and its APIs: https://github.com/simdjson/simdjson/blob/master/doc/basics.md
|
||||
* Performance shows some more advanced scenarios and how to tune for them: https://github.com/simdjson/simdjson/blob/master/doc/performance.md
|
||||
* Contributing: https://github.com/simdjson/simdjson/blob/master/CONTRIBUTING.md
|
||||
* We follow the [JSON specification as described by RFC 8259](https://www.rfc-editor.org/rfc/rfc8259.txt) (T. Bray, 2017).
|
||||
|
||||
We do not make changes to simdjson without clearly identifiable benefits, which typically means either performance improvements, bug fixes or new features. Avoid bike-shedding: we all have opinions about how to write code, but we want to focus on what makes simdjson objectively better.
|
||||
|
||||
|
||||
**Is your feature request related to a problem? Please describe.**
|
||||
A clear and concise description of what the problem is. Ex. I'm always frustrated when [...]
|
||||
|
||||
**Describe the solution you'd like**
|
||||
A clear and concise description of what you want to happen.
|
||||
|
||||
Please provide a clear rationale for the feature. Be advised that simdjson is a community-based project: you should consider providing help.
|
||||
|
||||
**Describe alternatives you've considered**
|
||||
A clear and concise description of any alternative solutions or features you've considered.
|
||||
|
||||
**Additional context**
|
||||
Add any other context or screenshots about the feature request here.
|
||||
|
||||
** Are you willing to contribute code or documentation toward this new feature? **
|
||||
If you plan to contribute to simdjson, please read our
|
||||
* CONTRIBUTING guide: https://github.com/simdjson/simdjson/blob/master/CONTRIBUTING.md and our
|
||||
* HACKING guide: https://github.com/simdjson/simdjson/blob/master/HACKING.md
|
||||
@@ -0,0 +1,32 @@
|
||||
---
|
||||
name: Standard issue template
|
||||
about: Issue
|
||||
title: ''
|
||||
labels: ''
|
||||
assignees: ''
|
||||
|
||||
---
|
||||
|
||||
Before submitting an issue, please ensure that you have read the documentation:
|
||||
|
||||
* Basics is an overview of how to use simdjson and its APIs: https://github.com/simdjson/simdjson/blob/master/doc/basics.md
|
||||
* Performance shows some more advanced scenarios and how to tune for them: https://github.com/simdjson/simdjson/blob/master/doc/performance.md
|
||||
* Contributing: https://github.com/simdjson/simdjson/blob/master/CONTRIBUTING.md
|
||||
* We follow the [JSON specification as described by RFC 8259](https://www.rfc-editor.org/rfc/rfc8259.txt) (T. Bray, 2017).
|
||||
|
||||
We do not make changes to simdjson without clearly identifiable benefits, which typically means either performance improvements, bug fixes or new features. Avoid bike-shedding: we all have opinions about how to write code, but we want to focus on what makes simdjson objectively better.
|
||||
|
||||
Is your issue:
|
||||
|
||||
1. A bug report? If so, please point at a reproducible test. Indicate whether you are willing or able to provide a bug fix as a pull request.
|
||||
|
||||
2. A build issue? If so, provide all possible details regarding your system configuration. If we cannot reproduce your issue, we cannot fix it.
|
||||
|
||||
3. A feature request? Please provide a clear rationale for the feature. Be advised that simdjson is a community-based project: you should consider providing help.
|
||||
|
||||
4. A documentation issue? Can you suggest an improvement?
|
||||
|
||||
|
||||
If you plan to contribute to simdjson, please read our
|
||||
* CONTRIBUTING guide: https://github.com/simdjson/simdjson/blob/master/CONTRIBUTING.md and our
|
||||
* HACKING guide: https://github.com/simdjson/simdjson/blob/master/HACKING.md
|
||||
@@ -0,0 +1,8 @@
|
||||
|
||||
|
||||
Our tests check whether you have introduced trailing white space. If such a test fails, please check the "artifacts button" above, which if you click it gives a link to a downloadable file to help you identify the issue. You can also run scripts/remove_trailing_whitespace.sh locally if you have a bash shell and the sed command available on your system.
|
||||
|
||||
If you plan to contribute to simdjson, please read our
|
||||
|
||||
CONTRIBUTING guide: https://github.com/simdjson/simdjson/blob/master/CONTRIBUTING.md and our
|
||||
HACKING guide: https://github.com/simdjson/simdjson/blob/master/HACKING.md
|
||||
@@ -0,0 +1,37 @@
|
||||
name: Alpine Linux
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
pull_request:
|
||||
branches:
|
||||
- master
|
||||
|
||||
jobs:
|
||||
ubuntu-build:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- uses: actions/cache@v2
|
||||
with:
|
||||
path: dependencies/.cache
|
||||
key: ${{ hashFiles('dependencies/CMakeLists.txt') }}
|
||||
- name: start docker
|
||||
run: |
|
||||
docker run -w /src -dit --name alpine -v $PWD:/src alpine:latest
|
||||
echo 'docker exec alpine "$@";' > ./alpine.sh
|
||||
chmod +x ./alpine.sh
|
||||
- name: install packages
|
||||
run: |
|
||||
./alpine.sh apk update
|
||||
./alpine.sh apk add build-base cmake g++ linux-headers git bash
|
||||
- name: cmake
|
||||
run: |
|
||||
./alpine.sh cmake -B build_for_alpine
|
||||
- name: build
|
||||
run: |
|
||||
./alpine.sh cmake --build build_for_alpine
|
||||
- name: test
|
||||
run: |
|
||||
./alpine.sh bash -c "cd build_for_alpine && ctest -E checkperf"
|
||||
@@ -0,0 +1,37 @@
|
||||
name: Detect trailing whitespace
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
branches:
|
||||
- master
|
||||
|
||||
jobs:
|
||||
whitespace:
|
||||
runs-on: ubuntu-20.04
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- name: Remove whitespace and check the diff
|
||||
run: |
|
||||
set -eu
|
||||
scripts/remove_trailing_whitespace.sh
|
||||
git diff >whitespace.patch
|
||||
cat whitespace.patch
|
||||
if [ $(wc -c <whitespace.patch) -ne 0 ] ; then
|
||||
echo " !!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!! "
|
||||
echo "You have trailing whitespace, please download the artifact"
|
||||
echo "and apply with git apply <whitespace.patch or"
|
||||
echo "run scripts/remove_trailing_whitespace.sh locally."
|
||||
echo " !!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!!! "
|
||||
exit 1
|
||||
else
|
||||
echo "no trailing whitespace found, good!"
|
||||
fi
|
||||
- name: Archive whitespace patch
|
||||
uses: actions/upload-artifact@v2
|
||||
if: always()
|
||||
with:
|
||||
name: whitespace-patch
|
||||
path: |
|
||||
whitespace.patch
|
||||
if-no-files-found: ignore
|
||||
|
||||
@@ -0,0 +1,161 @@
|
||||
name: Fuzz and run valgrind
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
pull_request:
|
||||
branches:
|
||||
- master
|
||||
schedule:
|
||||
- cron: 23 */8 * * *
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
# fuzzers that change behaviour with SIMDJSON_FORCE_IMPLEMENTATION
|
||||
defaultimplfuzzers: atpointer dump dump_raw_tape element minify parser print_json
|
||||
# fuzzers that loop over the implementations themselves, or don't need to switch.
|
||||
implfuzzers: implementations minifyimpl ndjson ondemand padded utf8
|
||||
implementations: haswell westmere fallback
|
||||
UBSAN_OPTIONS: halt_on_error=1
|
||||
MAXLEN: -max_len=4000
|
||||
CLANGVERSION: 11
|
||||
# which optimization level to use for the sanitizer build (see build_fuzzer.variants.sh)
|
||||
OPTLEVEL: -O3
|
||||
|
||||
steps:
|
||||
- name: Install packages necessary for building
|
||||
run: |
|
||||
sudo apt update
|
||||
sudo apt-get install --quiet ninja-build valgrind zip unzip
|
||||
wget https://apt.llvm.org/llvm.sh
|
||||
chmod +x llvm.sh
|
||||
sudo ./llvm.sh $CLANGVERSION
|
||||
|
||||
- uses: actions/checkout@v1
|
||||
|
||||
- uses: actions/cache@v2
|
||||
with:
|
||||
path: dependencies/.cache
|
||||
key: ${{ hashFiles('dependencies/CMakeLists.txt') }}
|
||||
|
||||
- name: Create and prepare the initial seed corpus
|
||||
run: |
|
||||
fuzz/build_corpus.sh
|
||||
mv corpus.zip seed_corpus.zip
|
||||
mkdir seedcorpus
|
||||
unzip -q -d seedcorpus seed_corpus.zip
|
||||
|
||||
- name: Download the corpus from the last run
|
||||
run: |
|
||||
wget --quiet https://dl.bintray.com/pauldreik/simdjson-fuzz-corpus/corpus/corpus.tar
|
||||
tar xf corpus.tar
|
||||
rm corpus.tar
|
||||
|
||||
- name: List clang versions
|
||||
run: |
|
||||
ls /usr/bin/clang*
|
||||
which clang++
|
||||
clang++ --version
|
||||
|
||||
- name: Build all the variants
|
||||
run: CLANGSUFFIX=-$CLANGVERSION fuzz/build_fuzzer_variants.sh
|
||||
|
||||
- name: Explore fast (release build, default implementation)
|
||||
run: |
|
||||
set -eux
|
||||
for fuzzer in $defaultimplfuzzers $implfuzzers; do
|
||||
mkdir -p out/$fuzzer # in case this is a new fuzzer, or corpus.tar is broken
|
||||
# get input from everyone else (corpus cross pollination)
|
||||
others=$(find out -type d -not -name $fuzzer -not -name out -not -name cmin)
|
||||
build-fast/fuzz/fuzz_$fuzzer out/$fuzzer $others seedcorpus -max_total_time=30 $MAXLEN
|
||||
done
|
||||
|
||||
- name: Fuzz default impl. fuzzers with sanitizer+asserts (good at detecting errors)
|
||||
run: |
|
||||
set -eux
|
||||
for fuzzer in $defaultimplfuzzers; do
|
||||
# get input from everyone else (corpus cross pollination)
|
||||
others=$(find out -type d -not -name $fuzzer -not -name out -not -name cmin)
|
||||
for implementation in $implementations; do
|
||||
export SIMDJSON_FORCE_IMPLEMENTATION=$implementation
|
||||
build-sanitizers$OPTLEVEL/fuzz/fuzz_$fuzzer out/$fuzzer $others seedcorpus -max_total_time=20 $MAXLEN
|
||||
done
|
||||
echo now have $(ls out/$fuzzer |wc -l) files in corpus
|
||||
done
|
||||
|
||||
- name: Fuzz differential impl. fuzzers with sanitizer+asserts (good at detecting errors)
|
||||
run: |
|
||||
set -eux
|
||||
for fuzzer in $implfuzzers; do
|
||||
# get input from everyone else (corpus cross pollination)
|
||||
others=$(find out -type d -not -name $fuzzer -not -name out -not -name cmin)
|
||||
build-sanitizers$OPTLEVEL/fuzz/fuzz_$fuzzer out/$fuzzer $others seedcorpus -max_total_time=20 $MAXLEN
|
||||
echo now have $(ls out/$fuzzer |wc -l) files in corpus
|
||||
done
|
||||
|
||||
- name: Minimize the corpus with the fast fuzzer on the default implementation
|
||||
run: |
|
||||
set -eux
|
||||
for fuzzer in $defaultimplfuzzers $implfuzzers; do
|
||||
mkdir -p out/cmin/$fuzzer
|
||||
# get input from everyone else (corpus cross pollination)
|
||||
others=$(find out -type d -not -name $fuzzer -not -name out -not -name cmin)
|
||||
build-fast/fuzz/fuzz_$fuzzer -merge=1 $MAXLEN out/cmin/$fuzzer out/$fuzzer $others seedcorpus
|
||||
rm -rf out/$fuzzer
|
||||
mv out/cmin/$fuzzer out/$fuzzer
|
||||
done
|
||||
|
||||
- name: Package the corpus into an artifact
|
||||
run: |
|
||||
for fuzzer in $defaultimplfuzzers $implfuzzers; do
|
||||
tar rf corpus.tar out/$fuzzer
|
||||
done
|
||||
|
||||
- name: Save the corpus as a github artifact
|
||||
uses: actions/upload-artifact@v2
|
||||
with:
|
||||
name: corpus
|
||||
path: corpus.tar
|
||||
|
||||
# This takes a subset of the minimized corpus and run it through valgrind. It is slow,
|
||||
# therefore take a "random" subset. The random selection is accomplished by sorting on filenames,
|
||||
# which are hashes of the content.
|
||||
- name: Run some of the minimized corpus through valgrind (replay build, default implementation)
|
||||
run: |
|
||||
for fuzzer in $defaultimplfuzzers $implfuzzers; do
|
||||
find out/$fuzzer -type f |sort|head -n200|xargs -n40 valgrind build-replay/fuzz/fuzz_$fuzzer 2>&1|tee valgrind-$fuzzer.txt
|
||||
done
|
||||
|
||||
- name: Compress the valgrind output
|
||||
run: tar cf valgrind.tar valgrind-*.txt
|
||||
|
||||
- name: Save valgrind output as a github artifact
|
||||
uses: actions/upload-artifact@v2
|
||||
if: always()
|
||||
with:
|
||||
name: valgrindresults
|
||||
path: valgrind.tar
|
||||
if-no-files-found: ignore
|
||||
|
||||
- name: Upload the corpus and results to bintray if we are on master
|
||||
if: ${{ github.event_name == 'schedule' }}
|
||||
run: |
|
||||
echo uploading each artifact twice, otherwise it will not be published
|
||||
curl -T corpus.tar -upauldreik:${{ secrets.bintrayApiKey }} https://api.bintray.com/content/pauldreik/simdjson-fuzz-corpus/corpus/0/corpus/corpus.tar";publish=1;override=1"
|
||||
curl -T corpus.tar -upauldreik:${{ secrets.bintrayApiKey }} https://api.bintray.com/content/pauldreik/simdjson-fuzz-corpus/corpus/0/corpus/corpus.tar";publish=1;override=1"
|
||||
curl -T valgrind.tar -upauldreik:${{ secrets.bintrayApiKey }} https://api.bintray.com/content/pauldreik/simdjson-fuzz-corpus/corpus/0/corpus/valgrind.tar";publish=1;override=1"
|
||||
curl -T valgrind.tar -upauldreik:${{ secrets.bintrayApiKey }} https://api.bintray.com/content/pauldreik/simdjson-fuzz-corpus/corpus/0/corpus/valgrind.tar";publish=1;override=1"
|
||||
|
||||
- name: Archive any crashes as an artifact
|
||||
uses: actions/upload-artifact@v2
|
||||
if: always()
|
||||
with:
|
||||
name: crashes
|
||||
path: |
|
||||
crash-*
|
||||
leak-*
|
||||
timeout-*
|
||||
if-no-files-found: ignore
|
||||
@@ -0,0 +1,65 @@
|
||||
name: MinGW32-CI
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
pull_request:
|
||||
branches:
|
||||
- master
|
||||
|
||||
# Important: scoop will either install 32-bit GCC or 64-bit GCC, not both.
|
||||
|
||||
# It is important to build static libraries because cmake is not smart enough under Windows/mingw to take care of the path. So
|
||||
# with a dynamic library, you could get failures due to the fact that the EXE can't find its DLL.
|
||||
|
||||
jobs:
|
||||
ci:
|
||||
name: windows-gcc
|
||||
runs-on: windows-2016
|
||||
|
||||
env:
|
||||
CMAKE_GENERATOR: Ninja # This is critical, try ' cmake -GNinja-DSIMDJSON_BUILD_STATIC=ON .. ' if using the command line
|
||||
CC: gcc
|
||||
CXX: g++
|
||||
|
||||
steps: # To reproduce what is below, start a powershell with administrative rights, using scoop *is* a good idea
|
||||
- uses: actions/checkout@v2
|
||||
|
||||
- uses: actions/cache@v2 # we cache the scoop setup with 32-bit GCC
|
||||
id: cache
|
||||
with:
|
||||
path: |
|
||||
C:\ProgramData\scoop
|
||||
key: scoop32 # static key: should be good forever
|
||||
|
||||
- uses: actions/cache@v2
|
||||
with:
|
||||
path: dependencies/.cache
|
||||
key: ${{ hashFiles('dependencies/CMakeLists.txt') }}
|
||||
|
||||
- name: Setup Windows # This should almost never run if the cache works.
|
||||
if: steps.cache.outputs.cache-hit != 'true'
|
||||
shell: powershell
|
||||
run: |
|
||||
Invoke-Expression (New-Object System.Net.WebClient).DownloadString('https://get.scoop.sh')
|
||||
scoop install sudo --global
|
||||
sudo scoop install git --global
|
||||
sudo scoop install ninja --global
|
||||
sudo scoop install cmake --global
|
||||
sudo scoop install gcc --arch 32bit --global
|
||||
$env:path
|
||||
Write-Host 'Everything has been installed, you are good!'
|
||||
- name: Build and Test 32-bit x86
|
||||
shell: powershell
|
||||
run: |
|
||||
$ENV:PATH="C:\ProgramData\scoop\shims;C:\ProgramData\scoop\apps\gcc\current\bin;C:\ProgramData\scoop\apps\ninja\current;$ENV:PATH"
|
||||
g++ --version
|
||||
cmake --version
|
||||
ninja --version
|
||||
git --version
|
||||
mkdir build32
|
||||
cd build32
|
||||
cmake -DSIMDJSON_BUILD_STATIC=ON -DSIMDJSON_COMPETITION=OFF -DSIMDJSON_GOOGLE_BENCHMARKS=OFF -DSIMDJSON_ENABLE_THREADS=OFF ..
|
||||
cmake --build . --target parse_many_test jsoncheck basictests ondemand_basictests numberparsingcheck stringparsingcheck errortests integer_tests pointercheck --verbose
|
||||
ctest -R "(parse_many_test|jsoncheck|basictests|stringparsingcheck|numberparsingcheck|errortests|integer_tests|pointercheck)" --output-on-failure
|
||||
@@ -0,0 +1,71 @@
|
||||
name: MinGW64-CI
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
pull_request:
|
||||
branches:
|
||||
- master
|
||||
|
||||
# Important: scoop will either install 32-bit GCC or 64-bit GCC, not both.
|
||||
|
||||
# It is important to build static libraries because cmake is not smart enough under Windows/mingw to take care of the path. So
|
||||
# with a dynamic library, you could get failures due to the fact that the EXE can't find its DLL.
|
||||
|
||||
jobs:
|
||||
ci:
|
||||
name: windows-gcc
|
||||
runs-on: windows-2016
|
||||
|
||||
env:
|
||||
CMAKE_GENERATOR: Ninja # This is critical, try ' cmake -GNinja-DSIMDJSON_BUILD_STATIC=ON .. ' if using the command line
|
||||
CC: gcc
|
||||
CXX: g++
|
||||
|
||||
steps: # To reproduce what is below, start a powershell with administrative rights, using scoop *is* a good idea
|
||||
- uses: actions/checkout@v2
|
||||
|
||||
- uses: actions/cache@v2 # we cache the scoop setup with 64-bit GCC
|
||||
id: cache
|
||||
with:
|
||||
path: |
|
||||
C:\ProgramData\scoop
|
||||
key: scoop64 # static key: should be good forever
|
||||
|
||||
- uses: actions/cache@v2
|
||||
with:
|
||||
path: dependencies/.cache
|
||||
key: ${{ hashFiles('dependencies/CMakeLists.txt') }}
|
||||
|
||||
- name: Setup Windows # This should almost never run if the cache works.
|
||||
if: steps.cache.outputs.cache-hit != 'true'
|
||||
shell: powershell
|
||||
run: |
|
||||
Invoke-Expression (New-Object System.Net.WebClient).DownloadString('https://get.scoop.sh')
|
||||
scoop install sudo --global
|
||||
sudo scoop install git --global
|
||||
sudo scoop install ninja --global
|
||||
sudo scoop install cmake --global
|
||||
sudo scoop install gcc --arch 64bit --global
|
||||
$env:path
|
||||
Write-Host 'Everything has been installed, you are good!'
|
||||
- name: Build and Test 64-bit x64
|
||||
shell: powershell
|
||||
run: |
|
||||
$ENV:PATH="C:\ProgramData\scoop\shims;C:\ProgramData\scoop\apps\gcc\current\bin;C:\ProgramData\scoop\apps\ninja\current;$ENV:PATH"
|
||||
g++ --version
|
||||
cmake --version
|
||||
ninja --version
|
||||
git --version
|
||||
mkdir build64
|
||||
cd build64
|
||||
cmake -DSIMDJSON_BUILD_STATIC=ON -DSIMDJSON_COMPETITION=OFF -DSIMDJSON_GOOGLE_BENCHMARKS=OFF -DSIMDJSON_ENABLE_THREADS=OFF ..
|
||||
cmake --build . --target parse_many_test jsoncheck basictests ondemand_basictests numberparsingcheck stringparsingcheck errortests integer_tests pointercheck --verbose
|
||||
ctest -R "(parse_many_test|jsoncheck|basictests|stringparsingcheck|numberparsingcheck|errortests|integer_tests|pointercheck)" --output-on-failure
|
||||
cd ..
|
||||
mkdir build64debug
|
||||
cd build64debug
|
||||
cmake -DCMAKE_BUILD_TYPE=Debug -DSIMDJSON_BUILD_STATIC=ON -DSIMDJSON_COMPETITION=OFF -DSIMDJSON_GOOGLE_BENCHMARKS=OFF -DSIMDJSON_ENABLE_THREADS=OFF ..
|
||||
cmake --build . --target parse_many_test jsoncheck basictests ondemand_basictests numberparsingcheck stringparsingcheck errortests integer_tests pointercheck --verbose
|
||||
ctest -R "(parse_many_test|jsoncheck|basictests|stringparsingcheck|numberparsingcheck|errortests|integer_tests|pointercheck)" --output-on-failure
|
||||
@@ -0,0 +1,54 @@
|
||||
name: MSYS2-CI
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
pull_request:
|
||||
branches:
|
||||
- master
|
||||
|
||||
jobs:
|
||||
windows-mingw:
|
||||
name: ${{ matrix.msystem }}
|
||||
runs-on: windows-latest
|
||||
defaults:
|
||||
run:
|
||||
shell: msys2 {0}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- msystem: "MINGW64"
|
||||
install: mingw-w64-x86_64-cmake mingw-w64-x86_64-ninja mingw-w64-x86_64-gcc
|
||||
type: Release
|
||||
- msystem: "MINGW32"
|
||||
install: mingw-w64-i686-cmake mingw-w64-i686-ninja mingw-w64-i686-gcc
|
||||
type: Release
|
||||
- msystem: "MINGW64"
|
||||
install: mingw-w64-x86_64-cmake mingw-w64-x86_64-ninja mingw-w64-x86_64-gcc
|
||||
type: Debug
|
||||
- msystem: "MINGW32"
|
||||
install: mingw-w64-i686-cmake mingw-w64-i686-ninja mingw-w64-i686-gcc
|
||||
type: Debug
|
||||
env:
|
||||
CMAKE_GENERATOR: Ninja
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- uses: actions/cache@v2
|
||||
with:
|
||||
path: dependencies/.cache
|
||||
key: ${{ hashFiles('dependencies/CMakeLists.txt') }}
|
||||
- uses: msys2/setup-msys2@v2
|
||||
with:
|
||||
update: true
|
||||
msystem: ${{ matrix.msystem }}
|
||||
install: ${{ matrix.install }}
|
||||
- name: Build and Test
|
||||
run: |
|
||||
mkdir build
|
||||
cd build
|
||||
cmake -DCMAKE_BUILD_TYPE=${{ matrix.type }} -DSIMDJSON_BUILD_STATIC=ON -DSIMDJSON_DO_NOT_USE_THREADS_NO_MATTER_WHAT=ON ..
|
||||
cmake --build . --verbose
|
||||
ctest -j4 --output-on-failure -E checkperf
|
||||
@@ -0,0 +1,26 @@
|
||||
name: Performance check on Ubuntu 18.04 CI (GCC 7)
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
pull_request:
|
||||
branches:
|
||||
- master
|
||||
|
||||
jobs:
|
||||
ubuntu-build:
|
||||
runs-on: ubuntu-18.04
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- uses: actions/cache@v2
|
||||
with:
|
||||
path: dependencies/.cache
|
||||
key: ${{ hashFiles('dependencies/CMakeLists.txt') }}
|
||||
- name: Use cmake
|
||||
run: |
|
||||
mkdir build &&
|
||||
cd build &&
|
||||
cmake -DSIMDJSON_GOOGLE_BENCHMARKS=ON -DSIMDJSON_BUILD_STATIC=ON -DCMAKE_INSTALL_PREFIX:PATH=destination .. &&
|
||||
cmake --build . --target checkperf &&
|
||||
ctest --output-on-failure -R checkperf ubuntu18-checkperf.yml
|
||||
@@ -0,0 +1,27 @@
|
||||
name: Ubuntu 18.04 CI (GCC 7) with Thread Sanitizer
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
pull_request:
|
||||
branches:
|
||||
- master
|
||||
|
||||
jobs:
|
||||
ubuntu-build:
|
||||
runs-on: ubuntu-18.04
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- uses: actions/cache@v2
|
||||
with:
|
||||
path: dependencies/.cache
|
||||
key: ${{ hashFiles('dependencies/CMakeLists.txt') }}
|
||||
- name: Use cmake
|
||||
run: |
|
||||
mkdir build &&
|
||||
cd build &&
|
||||
cmake -DSIMDJSON_SANITIZE_THREADS=ON .. &&
|
||||
cmake --build . --target document_stream_tests --target parse_many_test &&
|
||||
ctest --output-on-failure -R parse_many_test &&
|
||||
ctest --output-on-failure -R document_stream_tests
|
||||
@@ -0,0 +1,28 @@
|
||||
name: Ubuntu 18.04 CI (GCC 7)
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
pull_request:
|
||||
branches:
|
||||
- master
|
||||
|
||||
jobs:
|
||||
ubuntu-build:
|
||||
runs-on: ubuntu-18.04
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- uses: actions/cache@v2
|
||||
with:
|
||||
path: dependencies/.cache
|
||||
key: ${{ hashFiles('dependencies/CMakeLists.txt') }}
|
||||
- name: Use cmake
|
||||
run: |
|
||||
mkdir build &&
|
||||
cd build &&
|
||||
cmake -DSIMDJSON_GOOGLE_BENCHMARKS=ON -DSIMDJSON_BUILD_STATIC=ON -DCMAKE_INSTALL_PREFIX:PATH=destination .. &&
|
||||
cmake --build . &&
|
||||
ctest -j --output-on-failure -E checkperf &&
|
||||
make install &&
|
||||
echo -e '#include <simdjson.h>\nint main(int argc,char**argv) {simdjson::dom::parser parser;simdjson::dom::element tweets = parser.load(argv[1]); }' > tmp.cpp && c++ -Idestination/include -Ldestination/lib -std=c++17 -Wl,-rpath,destination/lib -o linkandrun tmp.cpp -lsimdjson && ./linkandrun jsonexamples/twitter.json
|
||||
@@ -0,0 +1,26 @@
|
||||
name: Performance check on Ubuntu 20.04 CI (GCC 9)
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
pull_request:
|
||||
branches:
|
||||
- master
|
||||
|
||||
jobs:
|
||||
ubuntu-build:
|
||||
runs-on: ubuntu-20.04
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- uses: actions/cache@v2
|
||||
with:
|
||||
path: dependencies/.cache
|
||||
key: ${{ hashFiles('dependencies/CMakeLists.txt') }}
|
||||
- name: Use cmake
|
||||
run: |
|
||||
mkdir build &&
|
||||
cd build &&
|
||||
cmake -DSIMDJSON_GOOGLE_BENCHMARKS=ON -DSIMDJSON_BUILD_STATIC=ON -DCMAKE_INSTALL_PREFIX:PATH=destination .. &&
|
||||
cmake --build . --target checkperf &&
|
||||
ctest --output-on-failure -R checkperf
|
||||
@@ -0,0 +1,27 @@
|
||||
name: Ubuntu 20.04 CI (GCC 9) with Thread Sanitizer
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
pull_request:
|
||||
branches:
|
||||
- master
|
||||
|
||||
jobs:
|
||||
ubuntu-build:
|
||||
runs-on: ubuntu-20.04
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- uses: actions/cache@v2
|
||||
with:
|
||||
path: dependencies/.cache
|
||||
key: ${{ hashFiles('dependencies/CMakeLists.txt') }}
|
||||
- name: Use cmake
|
||||
run: |
|
||||
mkdir build &&
|
||||
cd build &&
|
||||
cmake -DSIMDJSON_SANITIZE_THREADS=ON .. &&
|
||||
cmake --build . --target document_stream_tests --target parse_many_test &&
|
||||
ctest --output-on-failure -R parse_many_test &&
|
||||
ctest --output-on-failure -R document_stream_tests
|
||||
@@ -0,0 +1,28 @@
|
||||
name: Ubuntu 20.04 CI (GCC 9)
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
pull_request:
|
||||
branches:
|
||||
- master
|
||||
|
||||
jobs:
|
||||
ubuntu-build:
|
||||
runs-on: ubuntu-20.04
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- uses: actions/cache@v2
|
||||
with:
|
||||
path: dependencies/.cache
|
||||
key: ${{ hashFiles('dependencies/CMakeLists.txt') }}
|
||||
- name: Use cmake
|
||||
run: |
|
||||
mkdir build &&
|
||||
cd build &&
|
||||
cmake -DSIMDJSON_GOOGLE_BENCHMARKS=ON -DSIMDJSON_BUILD_STATIC=ON -DCMAKE_INSTALL_PREFIX:PATH=destination .. &&
|
||||
cmake --build . &&
|
||||
ctest -j --output-on-failure -E checkperf &&
|
||||
make install &&
|
||||
echo -e '#include <simdjson.h>\nint main(int argc,char**argv) {simdjson::dom::parser parser;simdjson::dom::element tweets = parser.load(argv[1]); }' > tmp.cpp && c++ -Idestination/include -Ldestination/lib -std=c++17 -Wl,-rpath,destination/lib -o linkandrun tmp.cpp -lsimdjson && ./linkandrun jsonexamples/twitter.json
|
||||
@@ -0,0 +1,36 @@
|
||||
name: VS16-CI
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
pull_request:
|
||||
branches:
|
||||
- master
|
||||
|
||||
jobs:
|
||||
ci:
|
||||
name: windows-vs16
|
||||
runs-on: windows-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- uses: actions/cache@v2
|
||||
with:
|
||||
path: dependencies/.cache
|
||||
key: ${{ hashFiles('dependencies/CMakeLists.txt') }}
|
||||
- name: 'Run CMake with VS16'
|
||||
uses: lukka/run-cmake@v2
|
||||
with:
|
||||
cmakeListsOrSettingsJson: CMakeListsTxtAdvanced
|
||||
cmakeListsTxtPath: '${{ github.workspace }}/CMakeLists.txt'
|
||||
buildDirectory: "${{ github.workspace }}/../../_temp/windows"
|
||||
cmakeBuildType: Release
|
||||
buildWithCMake: true
|
||||
cmakeGenerator: VS16Win64
|
||||
cmakeAppendedArgs: -DSIMDJSON_COMPETITION=OFF
|
||||
buildWithCMakeArgs: --config Release
|
||||
|
||||
- name: 'Run CTest'
|
||||
run: ctest -C Release -E checkperf --output-on-failure
|
||||
working-directory: "${{ github.workspace }}/../../_temp/windows"
|
||||
|
||||
@@ -0,0 +1,35 @@
|
||||
name: VS16-CLANG-CI
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
pull_request:
|
||||
branches:
|
||||
- master
|
||||
|
||||
jobs:
|
||||
ci:
|
||||
name: windows-vs16
|
||||
runs-on: windows-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- uses: actions/cache@v2
|
||||
with:
|
||||
path: dependencies/.cache
|
||||
key: ${{ hashFiles('dependencies/CMakeLists.txt') }}
|
||||
- name: 'Run CMake with VS16'
|
||||
uses: lukka/run-cmake@v2
|
||||
with:
|
||||
cmakeListsOrSettingsJson: CMakeListsTxtAdvanced
|
||||
cmakeListsTxtPath: '${{ github.workspace }}/CMakeLists.txt'
|
||||
buildDirectory: "${{ github.workspace }}/../../_temp/windows"
|
||||
cmakeBuildType: Release
|
||||
buildWithCMake: true
|
||||
cmakeGenerator: VS16Win64
|
||||
cmakeAppendedArgs: -T ClangCL -DSIMDJSON_COMPETITION=OFF -DSIMDJSON_BUILD_STATIC=ON
|
||||
buildWithCMakeArgs: --config Release
|
||||
|
||||
- name: 'Run CTest'
|
||||
run: ctest -C Release -E checkperf --output-on-failure
|
||||
working-directory: "${{ github.workspace }}/../../_temp/windows"
|
||||
@@ -0,0 +1,36 @@
|
||||
name: VS16-Ninja-CI
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
pull_request:
|
||||
branches:
|
||||
- master
|
||||
|
||||
jobs:
|
||||
ci:
|
||||
name: windows-vs16
|
||||
runs-on: windows-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- uses: actions/cache@v2
|
||||
with:
|
||||
path: dependencies/.cache
|
||||
key: ${{ hashFiles('dependencies/CMakeLists.txt') }}
|
||||
- name: 'Run CMake with VS16'
|
||||
uses: lukka/run-cmake@v2
|
||||
with:
|
||||
cmakeListsOrSettingsJson: CMakeListsTxtAdvanced
|
||||
cmakeListsTxtPath: '${{ github.workspace }}/CMakeLists.txt'
|
||||
buildDirectory: "${{ github.workspace }}/../../_temp/windows"
|
||||
cmakeBuildType: Release
|
||||
buildWithCMake: true
|
||||
cmakeGenerator: VS16Win64
|
||||
cmakeAppendedArgs: -G Ninja -DSIMDJSON_COMPETITION=OFF -DSIMDJSON_BUILD_STATIC=ON
|
||||
buildWithCMakeArgs: --config Release
|
||||
|
||||
- name: 'Run CTest'
|
||||
run: ctest -C Release -E checkperf --output-on-failure
|
||||
working-directory: "${{ github.workspace }}/../../_temp/windows"
|
||||
|
||||
+99
-1
@@ -1 +1,99 @@
|
||||
build/
|
||||
# eclipse project files
|
||||
.cproject
|
||||
.project
|
||||
.settings
|
||||
|
||||
# emacs temp files
|
||||
*~
|
||||
|
||||
# vim temp files
|
||||
.*.swp
|
||||
|
||||
# XCode
|
||||
^build/
|
||||
*.pbxuser
|
||||
!default.pbxuser
|
||||
*.mode1v3
|
||||
!default.mode1v3
|
||||
*.mode2v3
|
||||
!default.mode2v3
|
||||
*.perspectivev3
|
||||
!default.perspectivev3
|
||||
xcuserdata
|
||||
*.xccheckout
|
||||
*.moved-aside
|
||||
DerivedData
|
||||
*.hmap
|
||||
*.ipa
|
||||
*.xcuserstate
|
||||
*.DS_Store
|
||||
|
||||
# IDE specific folder for JetBrains IDEs
|
||||
.idea/
|
||||
cmake-build-debug/
|
||||
cmake-build-release/
|
||||
|
||||
# Visual Studio Code artifacts
|
||||
.vscode/*
|
||||
.history/
|
||||
|
||||
# Visual Studio artifacts
|
||||
/VS/
|
||||
|
||||
# C/C++ build outputs
|
||||
.build/
|
||||
bins
|
||||
gens
|
||||
libs
|
||||
objs
|
||||
|
||||
# C++ ignore from https://github.com/github/gitignore/blob/master/C%2B%2B.gitignore
|
||||
|
||||
# Prerequisites
|
||||
*.d
|
||||
|
||||
# Compiled Object files
|
||||
*.slo
|
||||
*.lo
|
||||
*.o
|
||||
*.obj
|
||||
|
||||
# Precompiled Headers
|
||||
*.gch
|
||||
*.pch
|
||||
|
||||
# Compiled Dynamic libraries
|
||||
*.so
|
||||
*.dylib
|
||||
*.dll
|
||||
|
||||
# Fortran module files
|
||||
*.mod
|
||||
*.smod
|
||||
|
||||
# Compiled Static libraries
|
||||
*.lai
|
||||
*.la
|
||||
*.a
|
||||
*.lib
|
||||
|
||||
# Executables
|
||||
*.exe
|
||||
*.out
|
||||
*.app
|
||||
|
||||
|
||||
# CMake files that may be specific to our installation
|
||||
|
||||
# Build outputs
|
||||
/build*/
|
||||
/visual_studio/
|
||||
|
||||
# Fuzzer outputs generated by instructions in fuzz/Fuzzing.md
|
||||
/corpus.zip
|
||||
/ossfuzz-out/
|
||||
/out/
|
||||
|
||||
# Generated docs
|
||||
/doc/api
|
||||
*.orig
|
||||
|
||||
-27
@@ -1,27 +0,0 @@
|
||||
[submodule "scalarvssimd/rapidjson"]
|
||||
path = dependencies/rapidjson
|
||||
url = https://github.com/Tencent/rapidjson.git
|
||||
[submodule "dependencies/sajson"]
|
||||
path = dependencies/sajson
|
||||
url = https://github.com/chadaustin/sajson.git
|
||||
[submodule "dependencies/json11"]
|
||||
path = dependencies/json11
|
||||
url = https://github.com/dropbox/json11.git
|
||||
[submodule "dependencies/fastjson"]
|
||||
path = dependencies/fastjson
|
||||
url = https://github.com/mikeando/fastjson.git
|
||||
[submodule "dependencies/gason"]
|
||||
path = dependencies/gason
|
||||
url = https://github.com/vivkin/gason.git
|
||||
[submodule "dependencies/ujson4c"]
|
||||
path = dependencies/ujson4c
|
||||
url = https://github.com/esnme/ujson4c.git
|
||||
[submodule "dependencies/jsmn"]
|
||||
path = dependencies/jsmn
|
||||
url = https://github.com/zserge/jsmn.git
|
||||
[submodule "dependencies/cJSON"]
|
||||
path = dependencies/cJSON
|
||||
url = https://github.com/DaveGamble/cJSON.git
|
||||
[submodule "dependencies/jsoncpp"]
|
||||
path = dependencies/jsoncpp
|
||||
url = https://github.com/open-source-parsers/jsoncpp.git
|
||||
+185
-16
@@ -1,19 +1,188 @@
|
||||
language: cpp
|
||||
sudo: false
|
||||
addons:
|
||||
apt:
|
||||
sources:
|
||||
- ubuntu-toolchain-r-test
|
||||
packages:
|
||||
- gcc-7
|
||||
- g++-7
|
||||
|
||||
branches:
|
||||
only:
|
||||
- master
|
||||
dist: bionic
|
||||
|
||||
script:
|
||||
- export CXX=g++-7
|
||||
- export CC=gcc-7
|
||||
- make
|
||||
- make test
|
||||
arch:
|
||||
- ppc64le
|
||||
|
||||
cache:
|
||||
directories:
|
||||
- $HOME/.dep_cache
|
||||
|
||||
env:
|
||||
global:
|
||||
- simdjson_DEPENDENCY_CACHE_DIR=$HOME/.dep_cache
|
||||
|
||||
matrix:
|
||||
include:
|
||||
- os: linux
|
||||
addons:
|
||||
apt:
|
||||
sources:
|
||||
- ubuntu-toolchain-r-test
|
||||
packages:
|
||||
- g++-8
|
||||
env:
|
||||
- COMPILER="CC=gcc-8 && CXX=g++-8"
|
||||
compiler: gcc-8
|
||||
|
||||
- os: linux
|
||||
addons:
|
||||
apt:
|
||||
sources:
|
||||
- ubuntu-toolchain-r-test
|
||||
packages:
|
||||
- g++-9
|
||||
env:
|
||||
- COMPILER="CC=gcc-9 && CXX=g++-9"
|
||||
compiler: gcc-9
|
||||
|
||||
- os: linux
|
||||
addons:
|
||||
apt:
|
||||
sources:
|
||||
- ubuntu-toolchain-r-test
|
||||
packages:
|
||||
- g++-10
|
||||
env:
|
||||
- COMPILER="CC=gcc-10 && CXX=g++-10"
|
||||
compiler: gcc-10
|
||||
|
||||
- os: linux
|
||||
addons:
|
||||
apt:
|
||||
sources:
|
||||
- ubuntu-toolchain-r-test
|
||||
packages:
|
||||
- g++-10
|
||||
env:
|
||||
- COMPILER="CC=gcc-10 && CXX=g++-10"
|
||||
- SANITIZE="on"
|
||||
compiler: gcc-10-sanitize
|
||||
|
||||
- os: linux
|
||||
addons:
|
||||
apt:
|
||||
sources:
|
||||
- ubuntu-toolchain-r-test
|
||||
packages:
|
||||
- g++-10
|
||||
env:
|
||||
- COMPILER="CC=gcc-10 && CXX=g++-10"
|
||||
- STATIC="on"
|
||||
compiler: gcc-10-static
|
||||
|
||||
- os: linux
|
||||
addons:
|
||||
apt:
|
||||
sources:
|
||||
- llvm-toolchain-bionic-6.0
|
||||
packages:
|
||||
- clang-6.0
|
||||
env:
|
||||
- COMPILER="CC=clang-6.0 && CXX=clang++-6.0"
|
||||
compiler: clang-6
|
||||
|
||||
- os: linux
|
||||
addons:
|
||||
apt:
|
||||
sources:
|
||||
- llvm-toolchain-bionic-7
|
||||
packages:
|
||||
- clang-7
|
||||
env:
|
||||
- COMPILER="CC=clang-7 && CXX=clang++-7"
|
||||
compiler: clang-7
|
||||
|
||||
- os: linux
|
||||
addons:
|
||||
apt:
|
||||
sources:
|
||||
- llvm-toolchain-bionic-8
|
||||
packages:
|
||||
- clang-8
|
||||
env:
|
||||
- COMPILER="CC=clang-8 && CXX=clang++-8"
|
||||
compiler: clang-8
|
||||
|
||||
- os: linux
|
||||
addons:
|
||||
apt:
|
||||
sources:
|
||||
- llvm-toolchain-bionic-9
|
||||
packages:
|
||||
- clang-9
|
||||
env:
|
||||
- COMPILER="CC=clang-9 && CXX=clang++-9"
|
||||
compiler: clang-9
|
||||
|
||||
- os: linux
|
||||
addons:
|
||||
apt:
|
||||
packages:
|
||||
- clang-10
|
||||
sources:
|
||||
- ubuntu-toolchain-r-test
|
||||
- sourceline: 'deb http://apt.llvm.org/bionic/ llvm-toolchain-bionic-10 main'
|
||||
key_url: 'https://apt.llvm.org/llvm-snapshot.gpg.key'
|
||||
env:
|
||||
- COMPILER="CC=clang-10 && CXX=clang++-10"
|
||||
compiler: clang-10
|
||||
|
||||
- os: linux
|
||||
addons:
|
||||
apt:
|
||||
packages:
|
||||
- clang-10
|
||||
sources:
|
||||
- ubuntu-toolchain-r-test
|
||||
- sourceline: 'deb http://apt.llvm.org/bionic/ llvm-toolchain-bionic-10 main'
|
||||
key_url: 'https://apt.llvm.org/llvm-snapshot.gpg.key'
|
||||
env:
|
||||
- COMPILER="CC=clang-10 && CXX=clang++-10"
|
||||
- STATIC="on"
|
||||
compiler: clang-10-static
|
||||
|
||||
- os: linux
|
||||
addons:
|
||||
apt:
|
||||
packages:
|
||||
- clang-10
|
||||
sources:
|
||||
- ubuntu-toolchain-r-test
|
||||
- sourceline: 'deb http://apt.llvm.org/bionic/ llvm-toolchain-bionic-10 main'
|
||||
key_url: 'https://apt.llvm.org/llvm-snapshot.gpg.key'
|
||||
env:
|
||||
- COMPILER="CC=clang-10 && CXX=clang++-10"
|
||||
- SANITIZE="on"
|
||||
compiler: clang-10-sanitize
|
||||
|
||||
before_install:
|
||||
- eval "${COMPILER}"
|
||||
|
||||
install:
|
||||
- wget -q -O - "https://raw.githubusercontent.com/simdjson/debian-ppa/master/key.gpg" | sudo apt-key add -
|
||||
- sudo apt-add-repository "deb https://raw.githubusercontent.com/simdjson/debian-ppa/master simdjson main"
|
||||
- sudo apt-get -qq update
|
||||
- sudo apt-get purge cmake cmake-data
|
||||
- sudo apt-get -t simdjson -y install cmake
|
||||
- export CMAKE_CXX_FLAGS="-maltivec -mcpu=power9 -mtune=power9"
|
||||
- export CMAKE_C_FLAGS="${CMAKE_CXX_FLAGS}"
|
||||
- export CMAKE_FLAGS="-DCMAKE_CXX_FLAGS=${CMAKE_CXX_FLAGS} -DCMAKE_C_FLAGS=${CMAKE_C_FLAGS} -DSIMDJSON_IMPLEMENTATION=ppc64;fallback";
|
||||
- if [[ "${SANITIZE}" == "on" ]]; then
|
||||
export CMAKE_FLAGS="${CMAKE_FLAGS} -DSIMDJSON_SANITIZE=ON";
|
||||
export ASAN_OPTIONS="detect_leaks=0";
|
||||
fi
|
||||
- if [[ "${STATIC}" == "on" ]]; then
|
||||
export CMAKE_FLAGS="${CMAKE_FLAGS} -DSIMDJSON_BUILD_STATIC=ON";
|
||||
fi
|
||||
- export CTEST_FLAGS="-j4 --output-on-failure -E checkperf"
|
||||
|
||||
script:
|
||||
- mkdir build
|
||||
- cd build
|
||||
- cmake $CMAKE_FLAGS ..
|
||||
- cmake --build . -- -j2
|
||||
- SIMDJSON_FORCE_IMPLEMENTATION=ppc64 ctest $CTEST_FLAGS -L per_implementation
|
||||
- SIMDJSON_FORCE_IMPLEMENTATION=fallback ctest $CTEST_FLAGS -L per_implementation
|
||||
- ctest $CTEST_FLAGS -LE "acceptance|per_implementation"
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
# List of authors for copyright purposes
|
||||
# List of authors for copyright purposes, in no particular order
|
||||
Daniel Lemire
|
||||
Geoff Langdale
|
||||
John Keiser
|
||||
|
||||
+66
-30
@@ -1,40 +1,76 @@
|
||||
cmake_minimum_required(VERSION 3.8...3.13)
|
||||
set(CMAKE_CXX_STANDARD 17)
|
||||
set(CMAKE_CXX_STANDARD_REQUIRED ON)
|
||||
set(CMAKE_MACOSX_RPATH OFF)
|
||||
if (NOT CMAKE_BUILD_TYPE)
|
||||
message(STATUS "No build type selected, default to Release")
|
||||
set(CMAKE_BUILD_TYPE Release CACHE STRING "Choose the type of build." FORCE)
|
||||
endif()
|
||||
cmake_minimum_required(VERSION 3.13)
|
||||
|
||||
project(simdjson)
|
||||
set(SIMDJSON_LIB_NAME simdjson)
|
||||
project(simdjson
|
||||
DESCRIPTION "Parsing gigabytes of JSON per second"
|
||||
LANGUAGES CXX C
|
||||
)
|
||||
|
||||
if(NOT MSVC)
|
||||
option(SIMDJSON_BUILD_STATIC "Build a static library" OFF) # turning it on disables the production of a dynamic library
|
||||
else()
|
||||
option(SIMDJSON_BUILD_STATIC "Build a static library" ON) # turning it on disables the production of a dynamic library
|
||||
endif()
|
||||
option(SIMDJSON_BUILD_LTO "Build library with Link Time Optimization" OFF)
|
||||
option(SIMDJSON_SANITIZE "Sanitize addresses" OFF)
|
||||
set(PROJECT_VERSION_MAJOR 0)
|
||||
set(PROJECT_VERSION_MINOR 7)
|
||||
set(PROJECT_VERSION_PATCH 1)
|
||||
set(SIMDJSON_SEMANTIC_VERSION "0.7.1" CACHE STRING "simdjson semantic version")
|
||||
set(SIMDJSON_LIB_VERSION "5.0.0" CACHE STRING "simdjson library version")
|
||||
set(SIMDJSON_LIB_SOVERSION "5" CACHE STRING "simdjson library soversion")
|
||||
set(SIMDJSON_GITHUB_REPOSITORY https://github.com/simdjson/simdjson)
|
||||
|
||||
set(CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/tools/cmake")
|
||||
include(GNUInstallDirs)
|
||||
include(cmake/simdjson-flags.cmake)
|
||||
include(cmake/simdjson-user-cmakecache.cmake)
|
||||
|
||||
find_package(CTargets)
|
||||
find_package(Options)
|
||||
find_package(LTO)
|
||||
|
||||
install(DIRECTORY include/${SIMDJSON_LIB_NAME} DESTINATION include)
|
||||
set (TEST_DATA_DIR "${CMAKE_CURRENT_SOURCE_DIR}/jsonchecker/")
|
||||
set (BENCHMARK_DATA_DIR "${CMAKE_CURRENT_SOURCE_DIR}/jsonexamples/")
|
||||
add_definitions(-DSIMDJSON_TEST_DATA_DIR="${TEST_DATA_DIR}")
|
||||
add_definitions(-DSIMDJSON_BENCHMARK_DATA_DIR="${TEST_DATA_DIR}")
|
||||
enable_testing()
|
||||
|
||||
if(SIMDJSON_JUST_LIBRARY)
|
||||
message( STATUS "Building just the library, omitting all tests, tools and benchmarks." )
|
||||
else(SIMDJSON_JUST_LIBRARY)
|
||||
# Setup tests
|
||||
enable_testing()
|
||||
add_subdirectory(jsonchecker)
|
||||
add_subdirectory(jsonexamples)
|
||||
add_library(test-data INTERFACE)
|
||||
target_link_libraries(test-data INTERFACE jsonchecker-data jsonchecker-minefield-data jsonexamples-data)
|
||||
endif(SIMDJSON_JUST_LIBRARY)
|
||||
|
||||
# Create the top level simdjson library (must be done at this level to use both src/ and include/
|
||||
# directories) and tools
|
||||
#
|
||||
add_subdirectory(include)
|
||||
add_subdirectory(src)
|
||||
add_subdirectory(tools)
|
||||
add_subdirectory(tests)
|
||||
add_subdirectory(benchmark)
|
||||
add_subdirectory(windows)
|
||||
if(NOT(SIMDJSON_JUST_LIBRARY))
|
||||
add_subdirectory(dependencies) ## This needs to be before tools because of cxxopts
|
||||
add_subdirectory(tools) ## This needs to be before tests because of cxxopts
|
||||
add_subdirectory(singleheader)
|
||||
endif()
|
||||
install(FILES singleheader/simdjson.h DESTINATION ${CMAKE_INSTALL_INCLUDEDIR})
|
||||
|
||||
#
|
||||
# Compile tools / tests / benchmarks
|
||||
#
|
||||
if(NOT(SIMDJSON_JUST_LIBRARY))
|
||||
add_subdirectory(tests)
|
||||
add_subdirectory(examples)
|
||||
add_subdirectory(benchmark)
|
||||
add_subdirectory(fuzz)
|
||||
endif()
|
||||
|
||||
#
|
||||
# Source files should be just ASCII
|
||||
#
|
||||
find_program(FIND find)
|
||||
find_program(FILE file)
|
||||
find_program(GREP grep)
|
||||
if((FIND) AND (FILE) AND (GREP))
|
||||
add_test(
|
||||
NAME "just_ascii"
|
||||
COMMAND sh -c "${FIND} include src windows tools singleheader tests examples benchmark -path benchmark/checkperf-reference -prune -name '*.h' -o -name '*.cpp' -type f -exec ${FILE} '{}' \; |${GREP} -v ASCII || exit 0 && exit 1"
|
||||
WORKING_DIRECTORY ${CMAKE_CURRENT_SOURCE_DIR}
|
||||
)
|
||||
endif()
|
||||
|
||||
|
||||
#
|
||||
# CPack
|
||||
#
|
||||
set(CPACK_PACKAGE_VENDOR "Daniel Lemire")
|
||||
set(CPACK_PACKAGE_CONTACT "lemire@gmail.com")
|
||||
set(CPACK_PACKAGE_DESCRIPTION_SUMMARY "Parsing gigabytes of JSON per second")
|
||||
|
||||
+100
@@ -0,0 +1,100 @@
|
||||
Contributing
|
||||
============
|
||||
|
||||
The simdjson library is an open project written in C++. Contributions are invited. Contributors
|
||||
agree to the project's license.
|
||||
|
||||
We have an extensive list of issues, and contributions toward any of these issues is invited.
|
||||
Contributions can take the form of code samples, better documentation or design ideas.
|
||||
|
||||
In particular, the following contributions are invited:
|
||||
|
||||
- The library is focused on performance. Well-documented performance optimization are invited.
|
||||
- Fixes to known or newly discovered bugs are always welcome. Typically, a bug fix should come with
|
||||
a test demonstrating that the bug has been fixed.
|
||||
- The simdjson library is advanced software and maintainability and flexibility are always a
|
||||
concern. Specific contributions to improve maintainability and flexibility are invited.
|
||||
|
||||
We discourage the following types of contributions:
|
||||
|
||||
- Code refactoring. We all have our preferences as to how code should be written, but unnecessary
|
||||
refactoring can waste time and introduce new bugs. If you believe that refactoring is needed, you
|
||||
first must explain how it helps in concrete terms. Does it improve the performance?
|
||||
- Applications of new language features for their own sake. Using advanced C++ language constructs
|
||||
is actually a negative as it may reduce portability (to old compilers, old standard libraries and
|
||||
systems) and reduce accessibility (to programmers that have not kept up), so it must be offsetted
|
||||
by clear gains like performance or maintainability. When in doubt, avoid advanced C++ features
|
||||
(beyond C++11).
|
||||
- Style formatting. In general, please abstain from reformatting code just to make it look prettier.
|
||||
Though code formatting is important, it can also be a waste of time if several contributors try to
|
||||
tweak the code base toward their own preference. Please do not introduce unneeded white-space
|
||||
changes.
|
||||
|
||||
In short, most code changes should either bring new features or better performance. We want to avoid unmotivated code changes.
|
||||
|
||||
|
||||
Specific rules
|
||||
----------
|
||||
|
||||
We have few hard rules, but we have some:
|
||||
|
||||
- Printing to standard output or standard error (`stderr`, `stdout`, `std::cerr`, `std::cout`) in the core library is forbidden. This follows from the [Writing R Extensions](https://cran.r-project.org/doc/manuals/R-exts.html) manual which states that "Compiled code should not write to stdout or stderr".
|
||||
- Calls to `abort()` are forbidden in the core library. This follows from the [Writing R Extensions](https://cran.r-project.org/doc/manuals/R-exts.html) manual which states that "Under no circumstances should your compiled code ever call abort or exit".
|
||||
- All source code files (.h, .cpp) must be ASCII.
|
||||
- All C macros introduced in public headers need to be prefixed with either `SIMDJSON_` or `simdjson_`.
|
||||
- We avoid trailing white space characters within lines. That is, your lines of code should not terminate with unnecessary spaces. Generally, please avoid making unnecessary changes to white-space characters when contributing code.
|
||||
|
||||
Tools, tests and benchmarks are not held to these same strict rules.
|
||||
|
||||
General Guidelines
|
||||
----------
|
||||
|
||||
Contributors are encouraged to :
|
||||
|
||||
- Document their changes. Though we do not enforce a rule regarding code comments, we prefer that non-trivial algorithms and techniques be somewhat documented in the code.
|
||||
- Follow as much as possible the existing code style. We do not enforce a specific code style, but we prefer consistency.
|
||||
- Modify as few lines of code as possible when working on an issue. The more lines you modify, the harder it is for your fellow human beings to understand what is going on.
|
||||
- Tools may report "problems" with the code, but we never delegate programming to tools: if there is a problem with the code, we need to understand it. Thus we will not "fix" code merely to please a static analyzer if we do not understand.
|
||||
- Provide tests for any new feature. We will not merge a new feature without tests.
|
||||
|
||||
Pull Requests
|
||||
--------------
|
||||
|
||||
Pull requests are always invited. However, we ask that you follow these guidelines:
|
||||
|
||||
- It is wiser to discuss your ideas first as part of an issue before you start coding. If you omit this step and code first, be prepare to have your code receive scrutiny and be dropped.
|
||||
- Users should provide a rationale for their changes. Does it improve performance? Does it add a feature? Does it improve maintainability? Does fix a bug? This must be explicitly stated as part of the pull request. Do not propose changes based on taste or intuition. We do not delegate programming to tools: that some tool suggested a code change is not reason enough to change the code.
|
||||
1. When your code improves performance, please document the gains with a benchmark using hard numbers.
|
||||
2. If your code fixes a bug, please be either fix a failing test, or propose a new test.
|
||||
3. Other types of changes must be clearly motivated. We openly discourage changes with no identifiable benefits.
|
||||
- Changes should be focused and minimal. You should change as few lines of code as possible. Please do not reformat or touch files needlessly.
|
||||
- New features must be accompanied of new tests, in general.
|
||||
- Your code should pass our continuous-integration tests. It is your responsability to ensure that your proposal pass the tests. We do not merge pull requests that would break our build.
|
||||
|
||||
If the benefits of your proposed code remain unclear, we may choose to discard your code: that is not an insult, we frequently discard our own code. We may also consider various alternatives and choose another path. Again, that is not an insult or a sign that you have wasted your time.
|
||||
|
||||
Style
|
||||
-----
|
||||
|
||||
Our formatting style is inspired by the LLVM style.
|
||||
The simdjson library is written using the snake case: when a variable or a function is a phrase, each space is replaced by an underscore character, and the first letter of each word written in lowercase. Compile-time constants are written entirely in uppercase with the same underscore convention.
|
||||
|
||||
Code of Conduct
|
||||
---------------
|
||||
|
||||
Though we do not have a formal code of conduct, we will not tolerate bullying, bigotry or
|
||||
intimidation. Everyone is welcome to contribute. If you have concerns, you can raise them privately with the core team members (e.g., D. Lemire, J. Keiser).
|
||||
|
||||
We welcome contributions from women and less represented groups. If you need help, please reach out.
|
||||
|
||||
Consider the following points when engaging with the project:
|
||||
|
||||
- We discourage arguments from authority: ideas are discusssed on their own merits and not based on who stated it.
|
||||
- Be mindful that what you may view as an aggression is maybe merely a difference of opinion or a misunderstanding.
|
||||
- Be mindful that a collection of small aggressions, even if mild in isolation, can become harmful.
|
||||
|
||||
Getting Started Hacking
|
||||
-----------------------
|
||||
|
||||
An overview of simdjson's directory structure, with pointers to architecture and design
|
||||
considerations and other helpful notes, can be found at [HACKING.md](HACKING.md).
|
||||
@@ -0,0 +1,40 @@
|
||||
# contributors (in no particular order)
|
||||
Thomas Navennec
|
||||
Kai Wolf
|
||||
Tyler Kennedy
|
||||
Frank Wessels
|
||||
George Fotopoulos
|
||||
Heinz N. Gies
|
||||
Emil Gedda
|
||||
Wojciech Muła
|
||||
Georgios Floros
|
||||
Dong Xie
|
||||
Nan Xiao
|
||||
Egor Bogatov
|
||||
Jinxi Wang
|
||||
Luiz Fernando Peres
|
||||
Wouter Bolsterlee
|
||||
Anish Karandikar
|
||||
Reini Urban
|
||||
Tom Dyson
|
||||
Ihor Dotsenko
|
||||
Alexey Milovidov
|
||||
Chang Liu
|
||||
Sunny Gleason
|
||||
John Keiser
|
||||
Zach Bjornson
|
||||
Vitaly Baranov
|
||||
Juho Lauri
|
||||
Michael Eisel
|
||||
Io Daza Dillon
|
||||
Paul Dreik
|
||||
Jeremie Piotte
|
||||
Matthew Wilson
|
||||
Dušan Jovanović
|
||||
Matjaž Ostroveršnik
|
||||
Nong Li
|
||||
Furkan Taşkale
|
||||
Brendan Knapp
|
||||
Danila Kutenin
|
||||
# if you have contributed to the project and your name does not
|
||||
# appear in this list, please let us know!
|
||||
+88
@@ -0,0 +1,88 @@
|
||||
###
|
||||
#
|
||||
# Though simdjson requires only commonly available compilers and tools, it can
|
||||
# be convenient to build it and test it inside a docker container: it makes it
|
||||
# possible to test and benchmark simdjson under even relatively out-of-date
|
||||
# Linux servers. It should also work under macOS and Windows, though not
|
||||
# at native speeds, maybe.
|
||||
#
|
||||
# Assuming that you have a working docker server, this file
|
||||
# allows you to build, test and benchmark simdjson.
|
||||
#
|
||||
# We build the library and associated files in the dockerbuild subdirectory.
|
||||
# It may be necessary to delete it before creating the image:
|
||||
#
|
||||
# rm -r -f dockerbuild
|
||||
#
|
||||
# The need to delete the directory has nothing to do with docker per se: it is
|
||||
# simply cleaner in CMake to start from a fresh directory. This is important: if you
|
||||
# reuse the same directory with different configurations, you may get broken builds.
|
||||
#
|
||||
#
|
||||
# Then you can build the image as follows:
|
||||
#
|
||||
# docker build -t simdjson --build-arg USER_ID=$(id -u) --build-arg GROUP_ID=$(id -g) .
|
||||
#
|
||||
# Please note that the image does not contain a copy of the code. However, the image will contain the
|
||||
# the compiler and the build system. This means that if you change the source code, after you have built
|
||||
# the image, you won't need to rebuild the image. In fact, unless you want to try a different compiler, you
|
||||
# do not need to ever rebuild the image, even if you do a lot of work on the source code.
|
||||
#
|
||||
# We specify the users to avoid having files owned by a privileged user (root) in our directory. Some
|
||||
# people like to run their machine as the "root" user. We do not think it is cool.
|
||||
#
|
||||
# Then you need to build the project:
|
||||
#
|
||||
# docker run -v $(pwd):/project:Z simdjson
|
||||
#
|
||||
# Should you change a source file, you may need to call this command again. Because the output
|
||||
# files are persistent between calls to this command (they reside in the dockerbuild directory),
|
||||
# this command can be fast.
|
||||
#
|
||||
# Next you can test it as follows:
|
||||
#
|
||||
# docker run -it -v $(pwd):/project:Z simdjson sh -c "cd dockerbuild && ctest . --output-on-failure -E checkperf"
|
||||
#
|
||||
# The run the complete tests requires you to have built all of simdjson.
|
||||
#
|
||||
# Building all of simdjson takes a long time. Instead, you can build just one target:
|
||||
#
|
||||
# docker run -it -v $(pwd):/project:Z simdjson sh -c "[ -d dockerbuild ] || mkdir dockerbuild && cd dockerbuild && cmake .. && cmake --build . --target parse"
|
||||
#
|
||||
# Note that it is safe to remove dockerbuild before call the previous command, as the repository gets rebuild. It is also possible, by changing the command, to use a different directory name.
|
||||
#
|
||||
# You can run performance tests:
|
||||
#
|
||||
# docker run -it --privileged -v $(pwd):/project:Z simdjson sh -c "cd dockerbuild && for i in ../jsonexamples/*.json; do echo \$i; ./benchmark/parse \$i; done"
|
||||
#
|
||||
# The "--privileged" is recommended so you can get performance counters under Linux.
|
||||
#
|
||||
# You can also grab a fresh copy of simdjson and rebuild it, to make comparisons:
|
||||
#
|
||||
# docker run -it -v $(pwd):/project:Z simdjson sh -c "git clone https://github.com/simdjson/simdjson.git && cd simdjson && mkdir build && cd build && cmake .. && cmake --build . --target parse "
|
||||
#
|
||||
# Then you can run comparisons:
|
||||
#
|
||||
# docker run -it --privileged -v $(pwd):/project:Z simdjson sh -c "for i in jsonexamples/*.json; do echo \$i; dockerbuild/benchmark/parse \$i| grep GB| head -n 1; simdjson/build/benchmark/parse \$i | grep GB |head -n 1; done"
|
||||
#
|
||||
####
|
||||
FROM ubuntu:20.10
|
||||
################
|
||||
# We would prefer to use the conan io images but they do not support 64-bit ARM? The small gcc images appear to
|
||||
# be broken on ARM.
|
||||
# Furthermore, we would not expect users to frequently rebuild the container, so using ubuntu is probably fine.
|
||||
###############
|
||||
ARG USER_ID
|
||||
ARG GROUP_ID
|
||||
RUN apt-get update -qq
|
||||
RUN DEBIAN_FRONTEND="noninteractive" apt-get -y install tzdata
|
||||
RUN apt-get install -y cmake g++ git
|
||||
RUN mkdir project
|
||||
|
||||
RUN addgroup --gid $GROUP_ID user; exit 0
|
||||
RUN adduser --disabled-password --gecos '' --uid $USER_ID --gid $GROUP_ID user; exit 0
|
||||
USER user
|
||||
RUN gcc --version
|
||||
WORKDIR /project
|
||||
|
||||
CMD ["sh","-c","[ -d dockerbuild ] || mkdir dockerbuild && cd dockerbuild && cmake .. && cmake --build . "]
|
||||
+293
@@ -0,0 +1,293 @@
|
||||
Hacking simdjson
|
||||
================
|
||||
|
||||
Here is wisdom about how to build, test and run simdjson from within the repository. This is mostly useful for people who plan to contribute simdjson, or maybe study the design.
|
||||
|
||||
If you plan to contribute to simdjson, please read our [CONTRIBUTING](https://github.com/simdjson/simdjson/blob/master/CONTRIBUTING.md) guide.
|
||||
|
||||
|
||||
Design notes
|
||||
------------------------------
|
||||
|
||||
The parser works in two stages:
|
||||
|
||||
- Stage 1. (Find marks) Identifies quickly structure elements, strings, and so forth. We validate UTF-8 encoding at that stage.
|
||||
- Stage 2. (Structure building) Involves constructing a "tree" of sort (materialized as a tape) to navigate through the data. Strings and numbers are parsed at this stage.
|
||||
|
||||
|
||||
The role of stage 1 is to identify pseudo-structural characters as quickly as possible. A character is pseudo-structural if and only if:
|
||||
|
||||
1. Not enclosed in quotes, AND
|
||||
2. Is a non-whitespace character, AND
|
||||
3. Its preceding character is either:
|
||||
(a) a structural character, OR
|
||||
(b) whitespace OR
|
||||
(c) the final quote in a string.
|
||||
|
||||
This helps as we redefine some new characters as pseudo-structural such as the characters 1, G, n in the following:
|
||||
|
||||
> { "foo" : 1.5, "bar" : 1.5 GEOFF_IS_A_DUMMY bla bla , "baz", null }
|
||||
|
||||
Stage 1 also does unicode validation.
|
||||
|
||||
Stage 2 handles all of the rest: number parsings, recognizing atoms like true, false, null, and so forth.
|
||||
|
||||
|
||||
Directory Structure and Source
|
||||
------------------------------
|
||||
|
||||
simdjson's source structure, from the top level, looks like this:
|
||||
|
||||
* **CMakeLists.txt:** The main build system.
|
||||
* **include:** User-facing declarations and inline definitions (most user-facing functions are inlined).
|
||||
* simdjson.h: A "main include" that includes files from include/simdjson/. This is equivalent to
|
||||
the distributed simdjson.h.
|
||||
* simdjson/*.h: Declarations for public simdjson classes and functions.
|
||||
* simdjson/*-inl.h: Definitions for public simdjson classes and functions.
|
||||
* **src:** The source files for non-inlined functionality (e.g. the architecture-specific parser
|
||||
implementations).
|
||||
* simdjson.cpp: A "main source" that includes all implementation files from src/. This is
|
||||
equivalent to the distributed simdjson.cpp.
|
||||
* arm64/|fallback/|haswell/|ppc64/|westmere/: Architecture-specific implementations. All functions are
|
||||
Each architecture defines its own namespace, e.g. simdjson::haswell.
|
||||
* generic/: Generic implementations of the simdjson parser. These files may be included and
|
||||
compiled multiple times, from whichever architectures use them. They assume they are already
|
||||
enclosed in a namespace, e.g.:
|
||||
```c++
|
||||
namespace simdjson {
|
||||
namespace haswell {
|
||||
#include "generic/stage1/json_structural_indexer.h"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
Other important files and directories:
|
||||
* **.drone.yml:** Definitions for Drone CI.
|
||||
* **.appveyor.yml:** Definitions for Appveyor CI (Windows).
|
||||
* **.circleci:** Definitions for Circle CI.
|
||||
* **.github/workflows:** Definitions for GitHub Actions (CI).
|
||||
* **singleheader:** Contains generated `simdjson.h` and `simdjson.cpp` that we release. The files `singleheader/simdjson.h` and `singleheader/simdjson.cpp` should never be edited by hand.
|
||||
* **singleheader/amalgamate.py:** Generates `singleheader/simdjson.h` and `singleheader/simdjson.cpp` for release (python script).
|
||||
* **benchmark:** This is where we do benchmarking. Benchmarking is core to every change we make; the
|
||||
cardinal rule is don't regress performance without knowing exactly why, and what you're trading
|
||||
for it. Many of our benchmarks are microbenchmarks. We are effectively doing controlled scientific experiments for the purpose of understanding what affects our performance. So we simplify as much as possible. We try to avoid irrelevant factors such as page faults, interrupts, unnnecessary system calls. We recommend checking the performance as follows:
|
||||
```bash
|
||||
mkdir build
|
||||
cd build
|
||||
cmake ..
|
||||
cmake --build . --config Release
|
||||
benchmark/parse ../jsonexamples/twitter.json
|
||||
```
|
||||
The last line becomes `./benchmark/Release/parse.exe ../jsonexample/twitter.json` under Windows. You may also use Google Benchmark:
|
||||
```bash
|
||||
mkdir build
|
||||
cd build
|
||||
cmake ..
|
||||
cmake --build . --target bench_parse_call --config Release
|
||||
./benchmark/bench_parse_call
|
||||
```
|
||||
The last line becomes `./benchmark/Release/bench_parse_call.exe` under Windows. Under Windows, you can also build with the clang compiler by adding `-T ClangCL` to the call to `cmake ..`: `cmake .. - TClangCL`.
|
||||
* **fuzz:** The source for fuzz testing. This lets us explore important edge and middle cases
|
||||
* **fuzz:** The source for fuzz testing. This lets us explore important edge and middle cases
|
||||
automatically, and is run in CI.
|
||||
* **jsonchecker:** A set of JSON files used to check different functionality of the parser.
|
||||
* **pass*.json:** Files that should pass validation.
|
||||
* **fail*.json:** Files that should fail validation.
|
||||
* **jsonchecker/minefield/y_*.json:** Files that should pass validation.
|
||||
* **jsonchecker/minefield/n_*.json:** Files that should fail validation.
|
||||
* **jsonexamples:** A wide spread of useful, real-world JSON files with different characteristics
|
||||
and sizes.
|
||||
* **test:** The tests are here. basictests.cpp and errortests.cpp are the primary ones.
|
||||
* **tools:** Source for executables that can be distributed with simdjson. Some examples:
|
||||
* `json2json mydoc.json` parses the document, constructs a model and then dumps back the result to standard output.
|
||||
* `json2json -d mydoc.json` parses the document, constructs a model and then dumps model (as a tape) to standard output. The tape format is described in the accompanying file `tape.md`.
|
||||
* `minify mydoc.json` minifies the JSON document, outputting the result to standard output. Minifying means to remove the unneeded white space characters.
|
||||
*`jsonpointer mydoc.json <jsonpath> <jsonpath> ... <jsonpath>` parses the document, constructs a model and then processes a series of [JSON Pointer paths](https://tools.ietf.org/html/rfc6901). The result is itself a JSON document.
|
||||
|
||||
|
||||
> **Don't modify the files in singleheader/ directly; these are automatically generated.**
|
||||
|
||||
|
||||
While simdjson distributes just two files from the singleheader/ directory, we *maintain* the code in
|
||||
multiple files under include/ and src/. The files include/simdjson.h and src/simdjson.cpp are the "spine" for
|
||||
these, and you can include them as if they were the corresponding singleheader/ files.
|
||||
|
||||
|
||||
|
||||
Runtime Dispatching
|
||||
--------------------
|
||||
|
||||
A key feature of simdjson is the ability to compile different processing kernels, optimized for specific instruction sets, and to select
|
||||
the most appropriate kernel at runtime. This ensures that users get the very best performance while still enabling simdjson to run everywhere.
|
||||
This technique is frequently called runtime dispatching. The simdjson achieves runtime dispatching entirely in C++: we do not assume
|
||||
that the user is building the code using CMake, for example.
|
||||
|
||||
To make runtime dispatching work, it is critical that the code be compiled for the lowest supported processor. In particular, you should
|
||||
not use flags such as -mavx2, /arch:AVX2 and so forth while compiling simdjson. When you do so, you allow the compiler to use advanced
|
||||
instructions. In turn, these advanced instructions present in the code may cause a runtime failure if the runtime processor does not
|
||||
support them. Even a simple loop, compiled with these flags, might generate binary code that only run on advanced processors.
|
||||
|
||||
So we compile simdjson for a generic processor. Our users should do the same if they want simdjson's runtime dispatch to work. It is important
|
||||
to understand that if runtime dispatching does not work, then simdjson will cause crashes on older processors. Of course, if a user chooses
|
||||
to compile their code for a specific instruction set (e.g., AVX2), they are responsible for the failures if they later run their code
|
||||
on a processor that does not support AVX2. Yet, if we were to entice these users to do so, we would share the blame: thus we carefully instruct
|
||||
users to compile their code in a generic way without doing anything to enable advanced instructions.
|
||||
|
||||
|
||||
We only use runtime dispatching on x64 (AMD/Intel) platforms, at the moment. On ARM processors, we would need a standard way to query, at runtime,
|
||||
the processor for its supported features. We do not know how to do so on ARM systems in general. Thankfully it is not yet a concern: 64-bit ARM
|
||||
processors are fairly uniform as far as the instruction sets they support.
|
||||
|
||||
|
||||
In all cases, simdjson uses advanced instructions by relying on "intrinsic functions": we do not write assembly code. The intrinsic functions
|
||||
are special functions that the compiler might recognize and translate into fast code. To make runtime dispatching work, we rely on the fact that
|
||||
the header providing these instructions
|
||||
(intrin.h under Visual Studio, x86intrin.h elsewhere) defines all of the intrinsic functions, including those that are not supported
|
||||
processor.
|
||||
|
||||
At this point, we are require to use one of two main strategies.
|
||||
|
||||
1. On POSIX systems, the main compilers (LLVM clang, GNU gcc) allow us to use any intrinsic function after including the header, but they fail to inline the resulting instruction if the target processor does not support them. Because we compile for a generic processor, we would not be able to use most intrinsic functions. Thankfully, more recent versions of these compilers allow us to flag a region of code with a specific target, so that we can compile only some of the code with support for advanced instructions. Thus in our C++, one might notice macros like `TARGET_HASWELL`. It is then our responsability, at runtime, to only run the regions of code (that we call kernels) matching the properties of the runtime processor. The benefit of this approach is that the compiler not only let us use intrinsic functions, but it can also optimize the rest of the code in the kernel with advanced instructions we enabled.
|
||||
|
||||
2. Under Visual Studio, the problem is somewhat simpler. Visual Studio will not only provide the intrinsic functions, but it will also allow us to use them. They will compile just fine. It is at runtime that they may cause a crash. So we do not need to mark regions of code for compilation toward advanced processors (e.g., with `TARGET_HASWELL` macros). The downside of the Visual Studio approach is that the compiler is not allowed to use advanced instructions others than those we specify. In principle, this means that Visual Studio has weaker optimization opportunities.
|
||||
|
||||
|
||||
|
||||
We also handle the special case where a user is compiling using LLVM clang under Windows, [using the Visual Studio toolchain](https://devblogs.microsoft.com/cppblog/clang-llvm-support-in-visual-studio/). If you compile with LLVM clang under Visual Studio, then the header files (intrin.h or x86intrin.h) no longer provides the intrinsic functions that are unsupported by the processor. This appears to be deliberate on the part of the LLVM engineers. With a few lines of code, we handle this scenario just like LLVM clang under a POSIX system, but forcing the inclusion of the specific headers, and rolling our own intrinsic function as needed.
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
Regenerating Single-Header Files
|
||||
---------------------------------------
|
||||
|
||||
The simdjson.h and simdjson.cpp files in the singleheader directory are not always up-to-date with the rest of the code; they are only ever
|
||||
systematically regenerated on releases. To ensure you have the latest code, you can regenerate them by running this at the top level:
|
||||
|
||||
```bash
|
||||
mkdir build
|
||||
cd build
|
||||
cmake ..
|
||||
cmake --build . # needed, because currently dependencies do not work fully for the amalgamate target
|
||||
cmake --build . --target amalgamate
|
||||
```
|
||||
|
||||
You need to have python3 installed on your system.
|
||||
|
||||
The amalgamator script `amalgamate.py` generates singleheader/simdjson.h by
|
||||
reading through include/simdjson.h, copy/pasting each header file into the amalgamated file at the
|
||||
point it gets included (but only once per header). singleheader/simdjson.cpp is generated from
|
||||
src/simdjson.cpp the same way, except files under generic/ may be included and copy/pasted multiple
|
||||
times.
|
||||
|
||||
### Usage (CMake on 64-bit platforms like Linux, FreeBSD or macOS)
|
||||
|
||||
Requirements: In addition to git, we require a recent version of CMake as well as bash.
|
||||
|
||||
1. On macOS, the easiest way to install cmake might be to use [brew](https://brew.sh) and then type
|
||||
```
|
||||
brew install cmake
|
||||
```
|
||||
2. Under Linux, you might be able to install CMake as follows:
|
||||
```
|
||||
apt-get update -qq
|
||||
apt-get install -y cmake
|
||||
```
|
||||
3. On FreeBSD, you might be able to install bash and CMake as follows:
|
||||
```
|
||||
pkg update -f
|
||||
pkg install bash
|
||||
pkg install cmake
|
||||
```
|
||||
|
||||
You need a recent compiler like clang or gcc. We recommend at least GNU GCC/G++ 7 or LLVM clang 6.
|
||||
|
||||
|
||||
Building: While in the project repository, do the following:
|
||||
|
||||
```
|
||||
mkdir build
|
||||
cd build
|
||||
cmake ..
|
||||
cmake --build .
|
||||
ctest
|
||||
```
|
||||
|
||||
CMake will build a library. By default, it builds a shared library (e.g., libsimdjson.so on Linux).
|
||||
|
||||
You can build a static library:
|
||||
|
||||
```
|
||||
mkdir buildstatic
|
||||
cd buildstatic
|
||||
cmake -DSIMDJSON_BUILD_STATIC=ON ..
|
||||
cmake --build .
|
||||
ctest
|
||||
```
|
||||
|
||||
In some cases, you may want to specify your compiler, especially if the default compiler on your system is too old. You need to tell cmake which compiler you wish to use by setting the CC and CXX variables. Under bash, you can do so with commands such as `export CC=gcc-7` and `export CXX=g++-7`. You can also do it as part of the `cmake` command: `cmake .. -DCMAKE_CXX_COMPILER=g++`. You may proceed as follows:
|
||||
|
||||
```
|
||||
brew install gcc@8
|
||||
mkdir build
|
||||
cd build
|
||||
export CXX=g++-8 CC=gcc-8
|
||||
cmake ..
|
||||
cmake --build .
|
||||
ctest
|
||||
```
|
||||
|
||||
If your compiler does not default on C++11 support or better you may get failing tests. If so, you may be able to exclude the failing tests by replacing `ctest` with `ctest -E "^quickstart$"`.
|
||||
|
||||
Note that the name of directory (`build`) is arbitrary, you can name it as you want (e.g., `buildgcc`) and you can have as many different such directories as you would like (one per configuration).
|
||||
|
||||
|
||||
|
||||
### Usage (CMake on 64-bit Windows using Visual Studio 2019)
|
||||
|
||||
We assume you have a common 64-bit Windows PC with at least Visual Studio 2019.
|
||||
|
||||
- Grab the simdjson code from GitHub, e.g., by cloning it using [GitHub Desktop](https://desktop.github.com/).
|
||||
- Install [CMake](https://cmake.org/download/). When you install it, make sure to ask that `cmake` be made available from the command line. Please choose a recent version of cmake.
|
||||
- Create a subdirectory within simdjson, such as `build`.
|
||||
- Using a shell, go to this newly created directory. You can start a shell directly from GitHub Desktop (Repository > Open in Command Prompt).
|
||||
- Type `cmake ..` in the shell while in the `build` repository.
|
||||
- This last command (`cmake ...`) created a Visual Studio solution file in the newly created directory (e.g., `simdjson.sln`). Open this file in Visual Studio. You should now be able to build the project and run the tests. For example, in the `Solution Explorer` window (available from the `View` menu), right-click `ALL_BUILD` and select `Build`. To test the code, still in the `Solution Explorer` window, select `RUN_TESTS` and select `Build`.
|
||||
|
||||
|
||||
Though having Visual Studio installed is necessary, one can build simdjson using only cmake commands:
|
||||
|
||||
- `mkdir build`
|
||||
- `cd build`
|
||||
- `cmake ..`
|
||||
- `cmake --build . -config Release`
|
||||
|
||||
|
||||
Furthermore, if you have installed LLVM clang on Windows, for example as a component of Visual Studio 2019, you can configure and build simdjson using LLVM clang on Windows using cmake:
|
||||
|
||||
|
||||
- `mkdir build`
|
||||
- `cd build`
|
||||
- `cmake .. -T ClangCL`
|
||||
- `cmake --build . -config Release`
|
||||
|
||||
|
||||
### Various References
|
||||
|
||||
- [How to implement atoi using SIMD?](https://stackoverflow.com/questions/35127060/how-to-implement-atoi-using-simd)
|
||||
- [Parsing JSON is a Minefield 💣](http://seriot.ch/parsing_json.php)
|
||||
- https://tools.ietf.org/html/rfc7159
|
||||
- http://rapidjson.org/md_doc_sax.html
|
||||
- https://github.com/Geal/parser_benchmarks/tree/master/json
|
||||
- Gron: A command line tool that makes JSON greppable https://news.ycombinator.com/item?id=16727665
|
||||
- GoogleGson https://github.com/google/gson
|
||||
- Jackson https://github.com/FasterXML/jackson
|
||||
- https://www.yelp.com/dataset_challenge
|
||||
- RapidJSON. http://rapidjson.org/
|
||||
|
||||
Inspiring links:
|
||||
|
||||
- https://auth0.com/blog/beating-json-performance-with-protobuf/
|
||||
- https://gist.github.com/shijuvar/25ad7de9505232c87034b8359543404a
|
||||
- https://github.com/frankmcsherry/blog/blob/master/posts/2018-02-11.md
|
||||
@@ -1,167 +0,0 @@
|
||||
|
||||
.SUFFIXES:
|
||||
#
|
||||
.SUFFIXES: .cpp .o .c .h
|
||||
|
||||
|
||||
.PHONY: clean cleandist
|
||||
COREDEPSINCLUDE = -Idependencies/rapidjson/include -Idependencies/sajson/include -Idependencies/cJSON -Idependencies/jsmn
|
||||
EXTRADEPSINCLUDE = -Idependencies/jsoncppdist -Idependencies/json11 -Idependencies/fastjson/src -Idependencies/fastjson/include -Idependencies/gason/src -Idependencies/ujson4c/3rdparty -Idependencies/ujson4c/src
|
||||
CXXFLAGS = -std=c++17 -march=native -Wall -Wextra -Wshadow -Iinclude -Ibenchmark/linux
|
||||
CFLAGS = -march=native -Idependencies/ujson4c/3rdparty -Idependencies/ujson4c/src
|
||||
ifeq ($(SANITIZE),1)
|
||||
CXXFLAGS += -g3 -O0 -fsanitize=address -fno-omit-frame-pointer -fsanitize=undefined
|
||||
CFLAGS += -g3 -O0 -fsanitize=address -fno-omit-frame-pointer -fsanitize=undefined
|
||||
else
|
||||
ifeq ($(DEBUG),1)
|
||||
CXXFLAGS += -g3 -O0
|
||||
CFLAGS += -g3 -O0
|
||||
else
|
||||
CXXFLAGS += -O3
|
||||
CFLAGS += -O3
|
||||
endif
|
||||
endif
|
||||
|
||||
MAINEXECUTABLES=parse minify json2json jsonstats statisticalmodel
|
||||
TESTEXECUTABLES=jsoncheck numberparsingcheck stringparsingcheck
|
||||
COMPARISONEXECUTABLES=minifiercompetition parsingcompetition parseandstatcompetition distinctuseridcompetition allparserscheckfile allparsingcompetition
|
||||
SUPPLEMENTARYEXECUTABLES=parse_noutf8validation parse_nonumberparsing parse_nostringparsing
|
||||
|
||||
HEADERS= include/simdjson/simdutf8check.h include/simdjson/stringparsing.h include/simdjson/numberparsing.h include/simdjson/jsonparser.h include/simdjson/common_defs.h include/simdjson/jsonioutil.h benchmark/benchmark.h benchmark/linux/linux-perf-events.h include/simdjson/parsedjson.h include/simdjson/stage1_find_marks.h include/simdjson/stage2_build_tape.h include/simdjson/jsoncharutils.h include/simdjson/jsonformatutils.h
|
||||
LIBFILES=src/jsonioutil.cpp src/jsonparser.cpp src/stage1_find_marks.cpp src/stage2_build_tape.cpp src/parsedjson.cpp src/parsedjsoniterator.cpp
|
||||
MINIFIERHEADERS=include/simdjson/jsonminifier.h include/simdjson/simdprune_tables.h
|
||||
MINIFIERLIBFILES=src/jsonminifier.cpp
|
||||
|
||||
|
||||
RAPIDJSON_INCLUDE:=dependencies/rapidjson/include
|
||||
SAJSON_INCLUDE:=dependencies/sajson/include
|
||||
JSON11_INCLUDE:=dependencies/json11/json11.hpp
|
||||
FASTJSON_INCLUDE:=dependencies/include/fastjson/fastjson.h
|
||||
GASON_INCLUDE:=dependencies/gason/src/gason.h
|
||||
UJSON4C_INCLUDE:=dependencies/ujson4c/src/ujdecode.c
|
||||
CJSON_INCLUDE:=dependencies/cJSON/cJSON.h
|
||||
JSMN_INCLUDE:=dependencies/jsmn/jsmn.h
|
||||
|
||||
|
||||
LIBS=$(RAPIDJSON_INCLUDE) $(SAJSON_INCLUDE) $(JSON11_INCLUDE) $(FASTJSON_INCLUDE) $(GASON_INCLUDE) $(UJSON4C_INCLUDE) $(CJSON_INCLUDE) $(JSMN_INCLUDE)
|
||||
|
||||
EXTRAOBJECTS=ujdecode.o
|
||||
all: $(MAINEXECUTABLES)
|
||||
|
||||
competition: $(COMPARISONEXECUTABLES)
|
||||
|
||||
.PHONY: benchmark test
|
||||
|
||||
benchmark:
|
||||
bash ./scripts/parser.sh
|
||||
bash ./scripts/parseandstat.sh
|
||||
|
||||
test: jsoncheck numberparsingcheck stringparsingcheck
|
||||
./numberparsingcheck
|
||||
./stringparsingcheck
|
||||
./jsoncheck
|
||||
./scripts/testjson2json.sh
|
||||
@echo
|
||||
@tput setaf 2
|
||||
@echo "It looks like the code is good!"
|
||||
@tput sgr0
|
||||
|
||||
quiettest: jsoncheck numberparsingcheck stringparsingcheck
|
||||
./numberparsingcheck
|
||||
./stringparsingcheck
|
||||
./jsoncheck
|
||||
./scripts/testjson2json.sh
|
||||
|
||||
amalgamate:
|
||||
./amalgamation.sh
|
||||
|
||||
$(SAJSON_INCLUDE):
|
||||
git submodule update --init --recursive
|
||||
|
||||
$(RAPIDJSON_INCLUDE):
|
||||
git submodule update --init --recursive
|
||||
|
||||
$(JSON11_INCLUDE):
|
||||
git submodule update --init --recursive
|
||||
|
||||
$(FASTJSON_INCLUDE):
|
||||
git submodule update --init --recursive
|
||||
|
||||
$(GASON_INCLUDE):
|
||||
git submodule update --init --recursive
|
||||
|
||||
$(UJSON4C_INCLUDE):
|
||||
git submodule update --init --recursive
|
||||
|
||||
|
||||
|
||||
parse: benchmark/parse.cpp $(HEADERS) $(LIBFILES)
|
||||
$(CXX) $(CXXFLAGS) -o parse $(LIBFILES) benchmark/parse.cpp $(LIBFLAGS)
|
||||
|
||||
statisticalmodel: benchmark/statisticalmodel.cpp $(HEADERS) $(LIBFILES)
|
||||
$(CXX) $(CXXFLAGS) -o statisticalmodel $(LIBFILES) benchmark/statisticalmodel.cpp $(LIBFLAGS)
|
||||
|
||||
|
||||
parse_noutf8validation: benchmark/parse.cpp $(HEADERS) $(LIBFILES)
|
||||
$(CXX) $(CXXFLAGS) -o parse_noutf8validation -DSIMDJSON_SKIPUTF8VALIDATION $(LIBFILES) benchmark/parse.cpp $(LIBFLAGS)
|
||||
|
||||
parse_nonumberparsing: benchmark/parse.cpp $(HEADERS) $(LIBFILES)
|
||||
$(CXX) $(CXXFLAGS) -o parse_nonumberparsing -DSIMDJSON_SKIPNUMBERPARSING $(LIBFILES) benchmark/parse.cpp $(LIBFLAGS)
|
||||
|
||||
parse_nostringparsing: benchmark/parse.cpp $(HEADERS) $(LIBFILES)
|
||||
$(CXX) $(CXXFLAGS) -o parse_nostringparsing -DSIMDJSON_SKIPSTRINGPARSING $(LIBFILES) benchmark/parse.cpp $(LIBFLAGS)
|
||||
|
||||
|
||||
jsoncheck:tests/jsoncheck.cpp $(HEADERS) $(LIBFILES)
|
||||
$(CXX) $(CXXFLAGS) -o jsoncheck $(LIBFILES) tests/jsoncheck.cpp -I. $(LIBFLAGS)
|
||||
|
||||
numberparsingcheck:tests/numberparsingcheck.cpp $(HEADERS) $(LIBFILES)
|
||||
$(CXX) $(CXXFLAGS) -o numberparsingcheck tests/numberparsingcheck.cpp src/jsonioutil.cpp src/jsonparser.cpp src/stage1_find_marks.cpp src/parsedjson.cpp -I. $(LIBFLAGS) -DJSON_TEST_NUMBERS
|
||||
|
||||
|
||||
stringparsingcheck:tests/stringparsingcheck.cpp $(HEADERS) $(LIBFILES)
|
||||
$(CXX) $(CXXFLAGS) -o stringparsingcheck tests/stringparsingcheck.cpp src/jsonioutil.cpp src/jsonparser.cpp src/stage1_find_marks.cpp src/parsedjson.cpp -I. $(LIBFLAGS) -DJSON_TEST_STRINGS
|
||||
|
||||
|
||||
minifiercompetition: benchmark/minifiercompetition.cpp $(HEADERS) $(LIBS) $(MINIFIERHEADERS) $(LIBFILES) $(MINIFIERLIBFILES)
|
||||
$(CXX) $(CXXFLAGS) -o minifiercompetition $(LIBFILES) $(MINIFIERLIBFILES) benchmark/minifiercompetition.cpp -I. $(LIBFLAGS) $(COREDEPSINCLUDE)
|
||||
|
||||
minify: tools/minify.cpp $(HEADERS) $(MINIFIERHEADERS) $(LIBFILES) $(MINIFIERLIBFILES)
|
||||
$(CXX) $(CXXFLAGS) -o minify $(MINIFIERLIBFILES) $(LIBFILES) tools/minify.cpp -I.
|
||||
|
||||
json2json: tools/json2json.cpp $(HEADERS) $(LIBFILES)
|
||||
$(CXX) $(CXXFLAGS) -o json2json $ tools/json2json.cpp $(LIBFILES) -I.
|
||||
|
||||
jsonstats: tools/jsonstats.cpp $(HEADERS) $(LIBFILES)
|
||||
$(CXX) $(CXXFLAGS) -o jsonstats $ tools/jsonstats.cpp $(LIBFILES) -I.
|
||||
|
||||
ujdecode.o: $(UJSON4C_INCLUDE)
|
||||
$(CC) $(CFLAGS) -c dependencies/ujson4c/src/ujdecode.c
|
||||
|
||||
parseandstatcompetition: benchmark/parseandstatcompetition.cpp $(HEADERS) $(LIBFILES)
|
||||
$(CXX) $(CXXFLAGS) -o parseandstatcompetition $(LIBFILES) benchmark/parseandstatcompetition.cpp -I. $(LIBFLAGS) $(COREDEPSINCLUDE)
|
||||
|
||||
distinctuseridcompetition: benchmark/distinctuseridcompetition.cpp $(HEADERS) $(LIBFILES)
|
||||
$(CXX) $(CXXFLAGS) -o distinctuseridcompetition $(LIBFILES) benchmark/distinctuseridcompetition.cpp -I. $(LIBFLAGS) $(COREDEPSINCLUDE)
|
||||
|
||||
parsingcompetition: benchmark/parsingcompetition.cpp $(HEADERS) $(LIBFILES)
|
||||
$(CXX) $(CXXFLAGS) -o parsingcompetition $(LIBFILES) benchmark/parsingcompetition.cpp -I. $(LIBFLAGS) $(COREDEPSINCLUDE)
|
||||
|
||||
allparsingcompetition: benchmark/parsingcompetition.cpp $(HEADERS) $(LIBFILES) $(EXTRAOBJECTS) $(LIBS)
|
||||
$(CXX) $(CXXFLAGS) -o allparsingcompetition $(LIBFILES) benchmark/parsingcompetition.cpp $(EXTRAOBJECTS) -I. $(LIBFLAGS) $(COREDEPSINCLUDE) $(EXTRADEPSINCLUDE) -DALLPARSER
|
||||
|
||||
|
||||
allparserscheckfile: tests/allparserscheckfile.cpp $(HEADERS) $(LIBFILES) $(EXTRAOBJECTS) $(LIBS)
|
||||
$(CXX) $(CXXFLAGS) -o allparserscheckfile $(LIBFILES) tests/allparserscheckfile.cpp $(EXTRAOBJECTS) -I. $(LIBFLAGS) $(COREDEPSINCLUDE) $(EXTRADEPSINCLUDE)
|
||||
|
||||
|
||||
cppcheck:
|
||||
cppcheck --enable=all src/*.cpp benchmarks/*.cpp tests/*.cpp -Iinclude -I. -Ibenchmark/linux
|
||||
|
||||
everything: $(MAINEXECUTABLES) $(EXTRA_EXECUTABLES) $(TESTEXECUTABLES) $(COMPARISONEXECUTABLES) $(SUPPLEMENTARYEXECUTABLES)
|
||||
|
||||
clean:
|
||||
rm -f $(EXTRAOBJECTS) $(MAINEXECUTABLES) $(EXTRA_EXECUTABLES) $(TESTEXECUTABLES) $(COMPARISONEXECUTABLES) $(SUPPLEMENTARYEXECUTABLES)
|
||||
|
||||
cleandist:
|
||||
rm -f $(EXTRAOBJECTS) $(MAINEXECUTABLES) $(EXTRA_EXECUTABLES) $(TESTEXECUTABLES) $(COMPARISONEXECUTABLES) $(SUPPLEMENTARYEXECUTABLES)
|
||||
@@ -1,85 +0,0 @@
|
||||
# Notes on simdson
|
||||
|
||||
## Rationale:
|
||||
|
||||
The simdjson project serves two purposes:
|
||||
|
||||
1. It creates a useful library for parsing JSON data quickly.
|
||||
|
||||
2. It is a demonstration of the use of SIMD and pipelined programming techniques to perform a complex and irregular task.
|
||||
These techniques include the use of large registers and SIMD instructions to process large amounts of input data at once,
|
||||
to hold larger entities than can typically be held in a single General Purpose Register (GPR), and to perform operations
|
||||
that are not cheap to perform without use of a SIMD unit (for example table lookup using permute instructions).
|
||||
|
||||
The other key technique is that the system is designed to minimize the number of unpredictable branches that must be taken
|
||||
to perform the task. Modern architectures are both wide and deep (4-wide pipelines with ~14 stages are commonplace). A
|
||||
recent Intel Architecture processor, for example, can perform 3 256-bit SIMD operations or 2 512-bit SIMD operations per
|
||||
cycle as well as other operations on general purpose registers or with the load/store unit. An incorrectly predicted branch
|
||||
will clear this pipeline. While it is rare that a programmer can achieve the maximum throughput on a machine, a developer
|
||||
may be missing the opportunity to carry out 56 operations for each branch miss.
|
||||
|
||||
Many code-bases make use of SIMD and deeply pipelined, "non-branchy", processing for regular tasks. Numerical problems
|
||||
(e.g. "matrix multiply") or simple 'bulk search' tasks (e.g. "count all the occurrences of a given character in a text",
|
||||
"find the first occurrence of the string 'foo' in a text") frequently use this class of techniques. We are demonstrating
|
||||
that these techniques can be applied to much more complex and less regular task.
|
||||
|
||||
## Design:
|
||||
|
||||
### Stage 1: SIMD over bytes; bit vector processing over bytes.
|
||||
|
||||
The first stage of our processing must identify key points in our input: the 'structural characters' of JSON (curly and
|
||||
square braces, colon, and comma), the start and end of strings as delineated by double quote characters, other JSON 'atoms'
|
||||
that are not distinguishable by simple characters (constructs such as "true", "false", "null" and numbers), as well as
|
||||
discovering these characters and atoms in the presence of both quoting conventions and backslash escaping conventions.
|
||||
|
||||
As such we follow the broad outline of the construction of a structural index as set forth in the Mison paper [XXX]; first,
|
||||
the discovery of odd-length sequences of backslash characters (which will cause quote characters immediately following to
|
||||
be escaped and not serve their quoting role but instead be literal charaters), second, the discovery of quote pairs (which
|
||||
cause structural characters within the quote pairs to also be merely literal characters and have no function as structural
|
||||
characters), then finally the discovery of structural characters not contained without the quote pairs.
|
||||
|
||||
We depart from the Mison paper in terms of method and overall design. In terms of method, the Mison paper uses iteration
|
||||
over bit vectors to discover backslash sequences and quote pairs; we introduce branch-free techniques to discover both of
|
||||
these properties.
|
||||
|
||||
We also make use of our ability to quickly detect whitespace in this early stage. We can use another bit-vector based
|
||||
transformation to discover locations in our data that follow a structural character or quote or whitespace and are not whitespace. Excluding locations within strings, and the structural characters we have already discovered,
|
||||
these locations are the only place that we can expect to see the starts of the JSON 'atoms'. These locations are thus
|
||||
treated as 'structural' ('pseudo-structural characters').
|
||||
|
||||
This stage involves either SIMD processing over out bytes or the manipulation of bit arrays that have 1 bit corresponding
|
||||
to 1 byte of input. As such, it can be quite inefficient for some inputs - it is possible to observe dozens of operations
|
||||
taking place to discover that there are in fact no odd-numbered sequences of backslashes or quotes in a given block of
|
||||
input. However, this inefficiency on such inputs is balanced by the fact that it costs no more to run this code over
|
||||
complex structured input, and the alternatives would generally involve running a number of unpredictable branches (for
|
||||
example, the loop branches in Mison that iterate over bit vectors).
|
||||
|
||||
### Stage 2: The transition from "SIMD over bytes" to "indices"
|
||||
|
||||
Our structural, pseudo-structural and other 'interesting' characters are relatively rare (TODO: quantify in detail -
|
||||
it's typically about 1 in 10). As such, continuing to process them as bit vectors will involve manipulating data structures
|
||||
that are relatively large as well as being fairly unpredictably spaced. We must transform these bitvectors of "interesting"
|
||||
locations into offsets.
|
||||
|
||||
Note that we can examine the character at the offset to discover what the original function of the item in the bitvector
|
||||
was. While the JSON structural characters and quotes are relatively self-explanatory (although working only with one offset
|
||||
at a time, we have lost the distinction between opening quotes and closing quotes, something that was available in Stage 1),
|
||||
it is a quirk of JSON that the legal atoms can all be distinguished from each other by their first character - 't' for
|
||||
'true', 'f' for 'false', 'n' for 'null' and the character class [0-9-] for numerical values.
|
||||
|
||||
Thus, the offset suffices, as long as we retain our original input.
|
||||
|
||||
Our current implementation involves a straightforward transformation of bitmaps to indices by use of the 'count trailing
|
||||
zeros' operation and the well-known operation to clear the lowest set bit. Note that this implementation introduces an
|
||||
unpredictable branch; unless there is a regular pattern in our bitmaps, we would expect to have at least one branch miss
|
||||
for each bitmap.
|
||||
|
||||
### Stage 3: Operation over indices
|
||||
|
||||
This now works over a dual structure.
|
||||
|
||||
1. The "state machine", whose role it is to validate the sequence of structural characters and ensure that the input is at least generally structured like valid JSON (after this stage, the only errors permissible should be malformed atoms and numbers). If and only if the "state machine" reached all accept states, then,
|
||||
|
||||
2. The "tape machine" will have produced valid output. The tape machine works blindly over characters writing records to tapes. These records create a lean but somewhat traversable linked structure that, for valid inputs, should represent what we need to know about the JSON input.
|
||||
|
||||
FIXME: a lot more detail is required on the operation of both these machines.
|
||||
@@ -1,461 +1,206 @@
|
||||
# simdjson : Parsing gigabytes of JSON per second
|
||||
[](https://cloud.drone.io/lemire/simdjson/)
|
||||
[](https://circleci.com/gh/lemire/simdjson)
|
||||
[](https://ci.appveyor.com/project/lemire/simdjson)
|
||||
[![][license img]][license]
|
||||
[](https://bugs.chromium.org/p/oss-fuzz/issues/list?sort=-opened&q=proj%3Asimdjson&can=2)
|
||||
/badge.svg)
|
||||
[/badge.svg)](https://simdjson.org/plots.html)
|
||||

|
||||

|
||||
[![][license img]][license] [](https://simdjson.org/api/0.7.0/index.html)
|
||||
|
||||
simdjson : Parsing gigabytes of JSON per second
|
||||
===============================================
|
||||
|
||||
<img src="images/logo.png" width="10%" style="float: right">
|
||||
JSON is everywhere on the Internet. Servers spend a *lot* of time parsing it. We need a fresh
|
||||
approach. The simdjson library uses commonly available SIMD instructions and microparallel algorithms
|
||||
to parse JSON 2.5x faster than RapidJSON and 25x faster than JSON for Modern C++.
|
||||
|
||||
* **Fast:** Over 2.5x faster than commonly used production-grade JSON parsers.
|
||||
* **Record Breaking Features:** Minify JSON at 6 GB/s, validate UTF-8 at 13 GB/s, NDJSON at 3.5 GB/s.
|
||||
* **Easy:** First-class, easy to use and carefully documented APIs.
|
||||
* **Beyond DOM:** Try the new On Demand API for twice the speed (>4GB/s).
|
||||
* **Strict:** Full JSON and UTF-8 validation, lossless parsing. Performance with no compromises.
|
||||
* **Automatic:** Selects a CPU-tailored parser at runtime. No configuration needed.
|
||||
* **Reliable:** From memory allocation to error handling, simdjson's design avoids surprises.
|
||||
* **Peer Reviewed:** Our research appears in venues like VLDB Journal, Software: Practice and Experience.
|
||||
|
||||
This library is part of the [Awesome Modern C++](https://awesomecpp.com) list.
|
||||
|
||||
Table of Contents
|
||||
-----------------
|
||||
|
||||
* [Quick Start](#quick-start)
|
||||
* [Documentation](#documentation)
|
||||
* [Performance results](#performance-results)
|
||||
* [Real-world usage](#real-world-usage)
|
||||
* [Bindings and Ports of simdjson](#bindings-and-ports-of-simdjson)
|
||||
* [About simdjson](#about-simdjson)
|
||||
* [Funding](#funding)
|
||||
* [Contributing to simdjson](#contributing-to-simdjson)
|
||||
* [License](#license)
|
||||
|
||||
Quick Start
|
||||
-----------
|
||||
|
||||
|
||||
## A C++ library to see how fast we can parse JSON with complete validation.
|
||||
The simdjson library is easily consumable with a single .h and .cpp file.
|
||||
|
||||
JSON documents are everywhere on the Internet. Servers spend a lot of time parsing these documents. We want to accelerate the parsing of JSON per se using commonly available SIMD instructions as much as possible while doing full validation (including character encoding).
|
||||
0. Prerequisites: `g++` (version 7 or better) or `clang++` (version 6 or better), and a 64-bit system with a command-line shell (e.g., Linux, macOS, freeBSD). We also support programming environments like Visual Studio and Xcode, but different steps are needed.
|
||||
1. Pull [simdjson.h](singleheader/simdjson.h) and [simdjson.cpp](singleheader/simdjson.cpp) into a directory, along with the sample file [twitter.json](jsonexamples/twitter.json).
|
||||
```
|
||||
wget https://raw.githubusercontent.com/simdjson/simdjson/master/singleheader/simdjson.h https://raw.githubusercontent.com/simdjson/simdjson/master/singleheader/simdjson.cpp https://raw.githubusercontent.com/simdjson/simdjson/master/jsonexamples/twitter.json
|
||||
```
|
||||
2. Create `quickstart.cpp`:
|
||||
|
||||
## Paper
|
||||
```c++
|
||||
#include "simdjson.h"
|
||||
int main(void) {
|
||||
simdjson::dom::parser parser;
|
||||
simdjson::dom::element tweets = parser.load("twitter.json");
|
||||
std::cout << tweets["search_metadata"]["count"] << " results." << std::endl;
|
||||
}
|
||||
```
|
||||
3. `c++ -o quickstart quickstart.cpp simdjson.cpp`
|
||||
4. `./quickstart`
|
||||
```
|
||||
100 results.
|
||||
```
|
||||
|
||||
A description of the design and implementation of simdjson appears at https://arxiv.org/abs/1902.08318 and an informal blog post providing some background and context is at https://branchfree.org/2019/02/25/paper-parsing-gigabytes-of-json-per-second/.
|
||||
## Some performance results
|
||||
Documentation
|
||||
-------------
|
||||
|
||||
We can use a quarter or fewer instructions than a state-of-the-art parser like RapidJSON, and half as many as sajson. To our knowledge, simdjson is the first fully-validating JSON parser to run at gigabytes per second on commodity processors.
|
||||
Usage documentation is available:
|
||||
|
||||
* [Basics](doc/basics.md) is an overview of how to use simdjson and its APIs.
|
||||
* [Performance](doc/performance.md) shows some more advanced scenarios and how to tune for them.
|
||||
* [Implementation Selection](doc/implementation-selection.md) describes runtime CPU detection and
|
||||
how you can work with it.
|
||||
* [API](https://simdjson.org/api/0.7.0/annotated.html) contains the automatically generated API documentation.
|
||||
|
||||
Performance results
|
||||
-------------------
|
||||
|
||||
The simdjson library uses three-quarters less instructions than state-of-the-art parser [RapidJSON](https://rapidjson.org) and
|
||||
fifty percent less than sajson. To our knowledge, simdjson is the first fully-validating JSON parser
|
||||
to run at [gigabytes per second](https://en.wikipedia.org/wiki/Gigabyte) (GB/s) on commodity processors. It can parse millions of JSON documents per second on a single core.
|
||||
|
||||
The following figure represents parsing speed in GB/s for parsing various files
|
||||
on an Intel Skylake processor (3.4 GHz) using the GNU GCC 9 compiler (with the -O3 flag).
|
||||
We compare against the best and fastest C++ libraries.
|
||||
The simdjson library offers full unicode ([UTF-8](https://en.wikipedia.org/wiki/UTF-8)) validation and exact
|
||||
number parsing. The RapidJSON library is tested in two modes: fast and
|
||||
exact number parsing. The sajson library offers fast (but not exact)
|
||||
number parsing and partial unicode validation. In this data set, the file
|
||||
sizes range from 65KB (github_events) all the way to 3.3GB (gsoc-2018).
|
||||
Many files are mostly made of numbers: canada, mesh.pretty, mesh, random
|
||||
and numbers: in such instances, we see lower JSON parsing speeds due to the
|
||||
high cost of number parsing. The simdjson library uses exact number parsing which
|
||||
is particular taxing.
|
||||
|
||||
<img src="doc/gbps.png" width="90%">
|
||||
|
||||
On a Skylake processor, the parsing speeds (in GB/s) of various processors on the twitter.json file are as follows, using again GNU GCC 9.1 (with the -O3 flag). The popular JSON for Modern C++ library is particularly slow: it obviously trades parsing speed for other desirable features.
|
||||
|
||||
On a Skylake processor, the parsing speeds (in GB/s) of various processors on the twitter.json file are as follows.
|
||||
| parser | GB/s |
|
||||
| ------------------------------------- | ---- |
|
||||
| simdjson | 2.5 |
|
||||
| RapidJSON UTF8-validation | 0.29 |
|
||||
| RapidJSON UTF8-valid., exact numbers | 0.28 |
|
||||
| RapidJSON insitu, UTF8-validation | 0.41 |
|
||||
| RapidJSON insitu, UTF8-valid., exact | 0.39 |
|
||||
| sajson (insitu, dynamic) | 0.62 |
|
||||
| sajson (insitu, static) | 0.88 |
|
||||
| dropbox | 0.13 |
|
||||
| fastjson | 0.27 |
|
||||
| gason | 0.59 |
|
||||
| ultrajson | 0.34 |
|
||||
| jsmn | 0.25 |
|
||||
| cJSON | 0.31 |
|
||||
| JSON for Modern C++ (nlohmann/json) | 0.11 |
|
||||
|
||||
| parser | GB/s |
|
||||
|---|---|
|
||||
| simdjson | 2.2 |
|
||||
| RapidJSON encoding-validation | 0.51|
|
||||
| RapidJSON encoding-validation, insitu | 0.71|
|
||||
| sajson (insitu, dynamic) | 0.70|
|
||||
| sajson (insitu, static) | 0.97|
|
||||
| dropbox | 0.14|
|
||||
| fastjson | 0.26|
|
||||
| gason | 0.85|
|
||||
| ultrajson | 0.42|
|
||||
| jsmn | 0.28|
|
||||
|cJSON | 0.34|
|
||||
|
||||
## Requirements
|
||||
The simdjson library offers high speed whether it processes tiny files (e.g., 300 bytes)
|
||||
or larger files (e.g., 3MB). The following plot presents parsing
|
||||
speed for [synthetic files over various sizes generated with a script](https://github.com/simdjson/simdjson_experiments_vldb2019/blob/master/experiments/growing/gen.py) on a 3.4 GHz Skylake processor (GNU GCC 9, -O3).
|
||||
<img src="doc/growing.png" width="90%">
|
||||
|
||||
- We support platforms like Linux or macOS, as well as Windows through Visual Studio 2017 or later.
|
||||
- A processor with AVX2 (i.e., Intel processors starting with the Haswell microarchitecture released 2013, and processors from AMD starting with the Ryzen)
|
||||
- A recent C++ compiler (e.g., GNU GCC or LLVM CLANG or Visual Studio 2017), we assume C++17. GNU GCC 7 or better or LLVM's clang 6 or better.
|
||||
- Some benchmark scripts assume bash and other common utilities, but they are optional.
|
||||
[All our experiments are reproducible](https://github.com/simdjson/simdjson_experiments_vldb2019).
|
||||
|
||||
## License
|
||||
|
||||
This code is made available under the Apache License 2.0.
|
||||
You can go beyond 4 GB/s with our new [On Demand API](https://github.com/simdjson/simdjson/blob/master/doc/ondemand.md).
|
||||
For NDJSON files, we can exceed 3 GB/s with [our multithreaded parsing functions](https://github.com/simdjson/simdjson/blob/master/doc/parse_many.md).
|
||||
|
||||
Under Windows, we build some tools using the windows/dirent_portable.h file (which is outside our library code): it under the liberal (business-friendly) MIT license.
|
||||
|
||||
## Code example
|
||||
|
||||
```C
|
||||
#include "simdjson/jsonparser.h"
|
||||
Real-world usage
|
||||
----------------
|
||||
|
||||
/...
|
||||
- [Microsoft FishStore](https://github.com/microsoft/FishStore)
|
||||
- [Yandex ClickHouse](https://github.com/yandex/ClickHouse)
|
||||
- [Clang Build Analyzer](https://github.com/aras-p/ClangBuildAnalyzer)
|
||||
- [Shopify HeapProfiler](https://github.com/Shopify/heap-profiler)
|
||||
|
||||
const char * filename = ... //
|
||||
If you are planning to use simdjson in a product, please work from one of our releases.
|
||||
|
||||
// use whatever means you want to get a string of your JSON document
|
||||
std::string_view p = get_corpus(filename);
|
||||
ParsedJson pj;
|
||||
pj.allocateCapacity(p.size()); // allocate memory for parsing up to p.size() bytes
|
||||
bool is_ok = json_parse(p, pj); // do the parsing, return false on error
|
||||
// parsing is done!
|
||||
// You can safely delete the string content
|
||||
free((void*)p.data());
|
||||
// the ParsedJson document can be used here
|
||||
// js can be reused with other json_parse calls.
|
||||
```
|
||||
Bindings and Ports of simdjson
|
||||
------------------------------
|
||||
|
||||
It is also possible to use a simpler API if you do not mind having the overhead
|
||||
of memory allocation with each new JSON document:
|
||||
|
||||
```C
|
||||
#include "simdjson/jsonparser.h"
|
||||
|
||||
/...
|
||||
|
||||
const char * filename = ... //
|
||||
std::string_view p = get_corpus(filename);
|
||||
ParsedJson pj = build_parsed_json(p); // do the parsing
|
||||
// you no longer need p at this point, can do aligned_free((void*)p.data())
|
||||
if( ! pj.isValid() ) {
|
||||
// something went wrong
|
||||
}
|
||||
```
|
||||
|
||||
## Usage: easy single-header version
|
||||
|
||||
See the "singleheader" repository for a single header version. See the included
|
||||
file "amalgamation_demo.cpp" for usage. This requires no specific build system: just
|
||||
copy the files in your project in your include path. You can then include them quite simply:
|
||||
|
||||
```C
|
||||
#include <iostream>
|
||||
#include "simdjson.h"
|
||||
#include "simdjson.cpp"
|
||||
int main(int argc, char *argv[]) {
|
||||
const char * filename = argv[1];
|
||||
std::string_view p = get_corpus(filename);
|
||||
ParsedJson pj = build_parsed_json(p); // do the parsing
|
||||
if( ! pj.isValid() ) {
|
||||
std::cout << "not valid" << std::endl;
|
||||
} else {
|
||||
std::cout << "valid" << std::endl;
|
||||
}
|
||||
return EXIT_SUCCESS;
|
||||
}
|
||||
```
|
||||
|
||||
Note: In some settings, it might be desirable to precompile `simdjson.cpp` instead of including it.
|
||||
|
||||
## Usage (old-school Makefile on platforms like Linux or macOS)
|
||||
|
||||
Requirements: recent clang or gcc, and make. We recommend at least GNU GCC/G++ 7 or LLVM clang 6. A system like Linux or macOS is expected.
|
||||
|
||||
To test:
|
||||
|
||||
```
|
||||
make
|
||||
make test
|
||||
```
|
||||
|
||||
|
||||
To run benchmarks:
|
||||
```
|
||||
make parse
|
||||
./parse jsonexamples/twitter.json
|
||||
```
|
||||
Under Linux, the `parse` command gives a detailed analysis of the performance counters.
|
||||
|
||||
To run comparative benchmarks (with other parsers):
|
||||
|
||||
```
|
||||
make benchmark
|
||||
```
|
||||
|
||||
## Usage (CMake on platforms like Linux or macOS)
|
||||
|
||||
Requirements: We require a recent version of cmake. On macOS, the easiest way to install cmake might be to use [brew](https://brew.sh) and then type
|
||||
|
||||
```
|
||||
brew install cmake
|
||||
```
|
||||
|
||||
There is an [equivalent brew on Linux which works the same way as well](https://linuxbrew.sh).
|
||||
|
||||
You need a recent compiler like clang or gcc. We recommend at least GNU GCC/G++ 7 or LLVM clang 6. For example, you can install a recent compiler with brew:
|
||||
|
||||
```
|
||||
brew install gcc@8
|
||||
```
|
||||
|
||||
Optional: You need to tell cmake which compiler you wish to use by setting the CC and CXX variables. Under bash, you can do so with commands such as ``export CC=gcc-7`` and ``export CXX=g++-7``.
|
||||
|
||||
|
||||
|
||||
Building: While in the project repository, do the following:
|
||||
|
||||
```
|
||||
mkdir build
|
||||
cd build
|
||||
cmake ..
|
||||
make
|
||||
make test
|
||||
```
|
||||
|
||||
CMake will build a library. By default, it builds a shared library (e.g., libsimdjson.so on Linux).
|
||||
|
||||
You can build a static library:
|
||||
|
||||
```
|
||||
mkdir buildstatic
|
||||
cd buildstatic
|
||||
cmake -DSIMDJSON_BUILD_STATIC=ON ..
|
||||
make
|
||||
make test
|
||||
```
|
||||
|
||||
|
||||
In some cases, you may want to specify your compiler, especially if the default compiler on your system is too old. You may proceed as follows:
|
||||
|
||||
```
|
||||
brew install gcc@8
|
||||
mkdir build
|
||||
cd build
|
||||
export CXX=g++-8 CC=gcc-8
|
||||
cmake ..
|
||||
make
|
||||
make test
|
||||
```
|
||||
|
||||
|
||||
## Usage (CMake on Windows using Visual Studio)
|
||||
|
||||
|
||||
We are assuming that you have a common Windows PC with at least Visual Studio 2017, and an x64 processor with AVX2 support (2013 Haswell or later).
|
||||
|
||||
- Grab the simdjson code from GitHub, e.g., by cloning it using [GitHub Desktop](https://desktop.github.com/).
|
||||
- Install [CMake](https://cmake.org/download/). When you install it, make sure to ask that ``cmake`` be made available from the command line. Please choose a recent version of cmake.
|
||||
- Create a subdirectory within simdjson, such as ``VisualStudio``.
|
||||
- Using a shell, go to this newly created directory.
|
||||
- Type ``cmake -DCMAKE_GENERATOR_PLATFORM=x64 ..`` in the shell while in the ``VisualStudio`` repository. (Alternatively, if you want to build a DLL, you may use the command line ``cmake -DCMAKE_GENERATOR_PLATFORM=x64 -DSIMDJSON_BUILD_STATIC=OFF ..``.)
|
||||
- This last command created a Visual Studio solution file in the newly created directory (e.g., ``simdjson.sln``). Open this file in Visual Studio. You should now be able to build the project and run the tests. For example, in the ``Solution Explorer`` window (available from the ``View`` menu), right-click ``ALL_BUILD`` and select ``Build``. To test the code, still in the ``Solution Explorer`` window, select ``RUN_TESTS`` and select ``Build``.
|
||||
|
||||
|
||||
## Tools
|
||||
|
||||
- `json2json mydoc.json` parses the document, constructs a model and then dumps back the result to standard output.
|
||||
- `json2json -d mydoc.json` parses the document, constructs a model and then dumps model (as a tape) to standard output. The tape format is described in the accompanying file `tape.md`.
|
||||
- `minify mydoc.json` minifies the JSON document, outputting the result to standard output. Minifying means to remove the unneeded white space characters.
|
||||
|
||||
## Scope
|
||||
|
||||
We provide a fast parser. It fully validates the input according to the various specifications.
|
||||
The parser builds a useful immutable (read-only) DOM (document-object model) which can be later accessed.
|
||||
|
||||
To simplify the engineering, we make some assumptions.
|
||||
|
||||
- We support UTF-8 (and thus ASCII), nothing else (no Latin, no UTF-16). We do not believe that this is a genuine limitation in the sense that we do not think that there is any serious application that needs to process JSON data without an ASCII or UTF-8 encoding.
|
||||
- We store strings as NULL terminated C strings. Thus we implicitly assume that you do not include a NULL character within your string, which is allowed technically speaking if you escape it (\u0000).
|
||||
- We assume AVX2 support which is available in all recent mainstream x86 processors produced by AMD and Intel. No support for non-x86 processors is included though it can be done. We plan to support ARM processors (help is invited).
|
||||
- In cases of failure, we just report a failure without any indication as to the nature of the problem. (This can be easily improved without affecting performance.)
|
||||
- As allowed by the specification, we allow repeated keys within an object (other parsers like sajson do the same).
|
||||
- Performance is optimized for JSON documents spanning at least a tens kilobytes up to many megabytes: the performance issues with having to parse many tiny JSON documents or one truly enormous JSON document are different.
|
||||
|
||||
*We do not aim to provide a general-purpose JSON library.* A library like RapidJSON offers much more than just parsing, it helps you generate JSON and offers various other convenient functions. We merely parse the document.
|
||||
|
||||
|
||||
## Features
|
||||
|
||||
- The input string is unmodified. (Parsers like sajson and RapidJSON use the input string as a buffer.)
|
||||
- We parse integers and floating-point numbers as separate types which allows us to support large 64-bit integers in [-9223372036854775808,9223372036854775808), like a Java `long` or a C/C++ `long long`. Among the parsers that differentiate between integers and floating-point numbers, not all support 64-bit integers. (For example, sajson rejects JSON files with integers larger than or equal to 2147483648. RapidJSON will parse a file containing an overly long integer like 18446744073709551616 as a floating-point number.) When we cannot represent exactly an integer as a signed 64-bit value, we reject the JSON document.
|
||||
- We do full UTF-8 validation as part of the parsing. (Parsers like fastjson, gason and dropbox json11 do not do UTF-8 validation.)
|
||||
- We fully validate the numbers. (Parsers like gason and ultranjson will accept `[0e+]` as valid JSON.)
|
||||
- We validate string content for unescaped characters. (Parsers like fastjson and ultrajson accept unescaped line breaks and tabs in strings.)
|
||||
|
||||
## Architecture
|
||||
|
||||
The parser works in two stages:
|
||||
|
||||
- Stage 1. (Find marks) Identifies quickly structure elements, strings, and so forth. We validate UTF-8 encoding at that stage.
|
||||
- Stage 2. (Structure building) Involves constructing a "tree" of sort (materialized as a tape) to navigate through the data. Strings and numbers are parsed at this stage.
|
||||
|
||||
|
||||
## Navigating the parsed document
|
||||
|
||||
Here is a code sample to dump back the parsed JSON to a string:
|
||||
|
||||
```c
|
||||
ParsedJson::iterator pjh(pj);
|
||||
if (!pjh.isOk()) {
|
||||
std::cerr << " Could not iterate parsed result. " << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
compute_dump(pj);
|
||||
//
|
||||
// where compute_dump is :
|
||||
|
||||
void compute_dump(ParsedJson::iterator &pjh) {
|
||||
if (pjh.is_object()) {
|
||||
std::cout << "{";
|
||||
if (pjh.down()) {
|
||||
pjh.print(std::cout); // must be a string
|
||||
std::cout << ":";
|
||||
pjh.next();
|
||||
compute_dump(pjh); // let us recurse
|
||||
while (pjh.next()) {
|
||||
std::cout << ",";
|
||||
pjh.print(std::cout);
|
||||
std::cout << ":";
|
||||
pjh.next();
|
||||
compute_dump(pjh); // let us recurse
|
||||
}
|
||||
pjh.up();
|
||||
}
|
||||
std::cout << "}";
|
||||
} else if (pjh.is_array()) {
|
||||
std::cout << "[";
|
||||
if (pjh.down()) {
|
||||
compute_dump(pjh); // let us recurse
|
||||
while (pjh.next()) {
|
||||
std::cout << ",";
|
||||
compute_dump(pjh); // let us recurse
|
||||
}
|
||||
pjh.up();
|
||||
}
|
||||
std::cout << "]";
|
||||
} else {
|
||||
pjh.print(std::cout); // just print the lone value
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
The following function will find all user.id integers:
|
||||
|
||||
```C
|
||||
void simdjson_traverse(std::vector<int64_t> &answer, ParsedJson::iterator &i) {
|
||||
switch (i.get_type()) {
|
||||
case '{':
|
||||
if (i.down()) {
|
||||
do {
|
||||
bool founduser = equals(i.get_string(), "user");
|
||||
i.next(); // move to value
|
||||
if (i.is_object()) {
|
||||
if (founduser && i.move_to_key("id")) {
|
||||
if (i.is_integer()) {
|
||||
answer.push_back(i.get_integer());
|
||||
}
|
||||
i.up();
|
||||
}
|
||||
simdjson_traverse(answer, i);
|
||||
} else if (i.is_array()) {
|
||||
simdjson_traverse(answer, i);
|
||||
}
|
||||
} while (i.next());
|
||||
i.up();
|
||||
}
|
||||
break;
|
||||
case '[':
|
||||
if (i.down()) {
|
||||
do {
|
||||
if (i.is_object_or_array()) {
|
||||
simdjson_traverse(answer, i);
|
||||
}
|
||||
} while (i.next());
|
||||
i.up();
|
||||
}
|
||||
break;
|
||||
case 'l':
|
||||
case 'd':
|
||||
case 'n':
|
||||
case 't':
|
||||
case 'f':
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
|
||||
## In-depth comparisons
|
||||
|
||||
If you want to see how a wide range of parsers validate a given JSON file:
|
||||
|
||||
```
|
||||
make allparserscheckfile
|
||||
./allparserscheckfile myfile.json
|
||||
```
|
||||
|
||||
For performance comparisons:
|
||||
|
||||
```
|
||||
make parsingcompetition
|
||||
./parsingcompetition myfile.json
|
||||
```
|
||||
|
||||
For broader comparisons:
|
||||
|
||||
```
|
||||
make allparsingcompetition
|
||||
./allparsingcompetition myfile.json
|
||||
```
|
||||
|
||||
## Other programming languages
|
||||
We distinguish between "bindings" (which just wrap the C++ code) and a port to another programming language (which reimplements everything).
|
||||
|
||||
- [ZippyJSON](https://github.com/michaeleisel/zippyjson): Swift bindings for the simdjson project.
|
||||
- [libpy_simdjson](https://github.com/gerrymanoim/libpy_simdjson/): high-speed Python bindings for simdjson using [libpy](https://github.com/quantopian/libpy).
|
||||
- [pysimdjson](https://github.com/TkTech/pysimdjson): Python bindings for the simdjson project.
|
||||
- [SimdJsonSharp](https://github.com/EgorBo/SimdJsonSharp): C# version for .NET Core
|
||||
- [simdjson-rs](https://github.com/simd-lite): Rust port.
|
||||
- [simdjson-rust](https://github.com/SunDoge/simdjson-rust): Rust wrapper (bindings).
|
||||
- [SimdJsonSharp](https://github.com/EgorBo/SimdJsonSharp): C# version for .NET Core (bindings and full port).
|
||||
- [simdjson_nodejs](https://github.com/luizperes/simdjson_nodejs): Node.js bindings for the simdjson project.
|
||||
- [simdjson_php](https://github.com/crazyxman/simdjson_php): PHP bindings for the simdjson project.
|
||||
- [simdjson_ruby](https://github.com/saka1/simdjson_ruby): Ruby bindings for the simdjson project.
|
||||
- [fast_jsonparser](https://github.com/anilmaurya/fast_jsonparser): Ruby bindings for the simdjson project.
|
||||
- [simdjson-go](https://github.com/minio/simdjson-go): Go port using Golang assembly.
|
||||
- [rcppsimdjson](https://github.com/eddelbuettel/rcppsimdjson): R bindings.
|
||||
- [simdjson_erlang](https://github.com/ChomperT/simdjson_erlang): erlang bindings.
|
||||
|
||||
|
||||
About simdjson
|
||||
--------------
|
||||
|
||||
## Various References
|
||||
The simdjson library takes advantage of modern microarchitectures, parallelizing with SIMD vector
|
||||
instructions, reducing branch misprediction, and reducing data dependency to take advantage of each
|
||||
CPU's multiple execution cores.
|
||||
|
||||
- [Google double-conv](https://github.com/google/double-conversion/)
|
||||
- [How to implement atoi using SIMD?](https://stackoverflow.com/questions/35127060/how-to-implement-atoi-using-simd)
|
||||
- [Parsing JSON is a Minefield 💣](http://seriot.ch/parsing_json.php)
|
||||
- https://tools.ietf.org/html/rfc7159
|
||||
- The Mison implementation in rust https://github.com/pikkr/pikkr
|
||||
- http://rapidjson.org/md_doc_sax.html
|
||||
- https://github.com/Geal/parser_benchmarks/tree/master/json
|
||||
- Gron: A command line tool that makes JSON greppable https://news.ycombinator.com/item?id=16727665
|
||||
- GoogleGson https://github.com/google/gson
|
||||
- Jackson https://github.com/FasterXML/jackson
|
||||
- https://www.yelp.com/dataset_challenge
|
||||
- RapidJSON. http://rapidjson.org/
|
||||
Some people [enjoy reading our paper](https://arxiv.org/abs/1902.08318): A description of the design
|
||||
and implementation of simdjson is in our research article:
|
||||
- Geoff Langdale, Daniel Lemire, [Parsing Gigabytes of JSON per Second](https://arxiv.org/abs/1902.08318), VLDB Journal 28 (6), 2019.
|
||||
|
||||
Inspiring links:
|
||||
- https://auth0.com/blog/beating-json-performance-with-protobuf/
|
||||
- https://gist.github.com/shijuvar/25ad7de9505232c87034b8359543404a
|
||||
- https://github.com/frankmcsherry/blog/blob/master/posts/2018-02-11.md
|
||||
We have an in-depth paper focused on the UTF-8 validation:
|
||||
|
||||
- John Keiser, Daniel Lemire, [Validating UTF-8 In Less Than One Instruction Per Byte](https://arxiv.org/abs/2010.03090), Software: Practice & Experience (to appear)
|
||||
|
||||
Validating UTF-8 takes no more than 0.7 cycles per byte:
|
||||
- https://github.com/lemire/fastvalidate-utf-8 https://lemire.me/blog/2018/05/16/validating-utf-8-strings-using-as-little-as-0-7-cycles-per-byte/
|
||||
We also have an informal [blog post providing some background and context](https://branchfree.org/2019/02/25/paper-parsing-gigabytes-of-json-per-second/).
|
||||
|
||||
For the video inclined, <br />
|
||||
[](http://www.youtube.com/watch?v=wlvKAT7SZIQ)<br />
|
||||
(it was the best voted talk, we're kinda proud of it).
|
||||
|
||||
## Remarks on JSON parsing
|
||||
Funding
|
||||
-------
|
||||
|
||||
- The JSON spec defines what a JSON parser is:
|
||||
> A JSON parser transforms a JSON text into another representation. A JSON parser MUST accept all texts that conform to the JSON grammar. A JSON parser MAY accept non-JSON forms or extensions. An implementation may set limits on the size of texts that it accepts. An implementation may set limits on the maximum depth of nesting. An implementation may set limits on the range and precision of numbers. An implementation may set limits on the length and character contents of strings.
|
||||
The work is supported by the Natural Sciences and Engineering Research Council of Canada under grant
|
||||
number RGPIN-2017-03910.
|
||||
|
||||
[license]: LICENSE
|
||||
[license img]: https://img.shields.io/badge/License-Apache%202-blue.svg
|
||||
|
||||
- JSON is not JavaScript:
|
||||
> All JSON is Javascript but NOT all Javascript is JSON. So {property:1} is invalid because property does not have double quotes around it. {'property':1} is also invalid, because it's single quoted while the only thing that can placate the JSON specification is double quoting. JSON is even fussy enough that {"property":.1} is invalid too, because you should have of course written {"property":0.1}. Also, don't even think about having comments or semicolons, you guessed it: they're invalid. (credit:https://github.com/elzr/vim-json)
|
||||
Contributing to simdjson
|
||||
------------------------
|
||||
|
||||
- The structural characters are:
|
||||
Head over to [CONTRIBUTING.md](CONTRIBUTING.md) for information on contributing to simdjson, and
|
||||
[HACKING.md](HACKING.md) for information on source, building, and architecture/design.
|
||||
|
||||
License
|
||||
-------
|
||||
|
||||
begin-array = [ left square bracket
|
||||
begin-object = { left curly bracket
|
||||
end-array = ] right square bracket
|
||||
end-object = } right curly bracket
|
||||
name-separator = : colon
|
||||
value-separator = , comma
|
||||
This code is made available under the [Apache License 2.0](https://www.apache.org/licenses/LICENSE-2.0.html).
|
||||
|
||||
Under Windows, we build some tools using the windows/dirent_portable.h file (which is outside our library code): it under the liberal (business-friendly) MIT license.
|
||||
|
||||
### Pseudo-structural elements
|
||||
|
||||
A character is pseudo-structural if and only if:
|
||||
|
||||
1. Not enclosed in quotes, AND
|
||||
2. Is a non-whitespace character, AND
|
||||
3. It's preceding character is either:
|
||||
(a) a structural character, OR
|
||||
(b) whitespace.
|
||||
|
||||
This helps as we redefine some new characters as pseudo-structural such as the characters 1, 1, G, n in the following:
|
||||
|
||||
> { "foo" : 1.5, "bar" : 1.5 GEOFF_IS_A_DUMMY bla bla , "baz", null }
|
||||
|
||||
|
||||
|
||||
|
||||
## Academic References
|
||||
|
||||
- T.Mühlbauer, W.Rödiger, R.Seilbeck, A.Reiser, A.Kemper, and T.Neumann. Instant loading for main memory databases. PVLDB, 6(14):1702–1713, 2013. (SIMD-based CSV parsing)
|
||||
- Mytkowicz, Todd, Madanlal Musuvathi, and Wolfram Schulte. "Data-parallel finite-state machines." ACM SIGARCH Computer Architecture News. Vol. 42. No. 1. ACM, 2014.
|
||||
- Lu, Yifan, et al. "Tree structured data processing on GPUs." Cloud Computing, Data Science & Engineering-Confluence, 2017 7th International Conference on. IEEE, 2017.
|
||||
- Sidhu, Reetinder. "High throughput, tree automata based XML processing using FPGAs." Field-Programmable Technology (FPT), 2013 International Conference on. IEEE, 2013.
|
||||
- Dai, Zefu, Nick Ni, and Jianwen Zhu. "A 1 cycle-per-byte XML parsing accelerator." Proceedings of the 18th annual ACM/SIGDA international symposium on Field programmable gate arrays. ACM, 2010.
|
||||
- Lin, Dan, et al. "Parabix: Boosting the efficiency of text processing on commodity processors." High Performance Computer Architecture (HPCA), 2012 IEEE 18th International Symposium on. IEEE, 2012. http://parabix.costar.sfu.ca/export/1783/docs/HPCA2012/final_ieee/final.pdf
|
||||
- Deshmukh, V. M., and G. R. Bamnote. "An empirical evaluation of optimization parameters in XML parsing for performance enhancement." Computer, Communication and Control (IC4), 2015 International Conference on. IEEE, 2015.
|
||||
- Moussalli, Roger, et al. "Efficient XML Path Filtering Using GPUs." ADMS@ VLDB. 2011.
|
||||
- Jianliang, Ma, et al. "Parallel speculative dom-based XML parser." High Performance Computing and Communication & 2012 IEEE 9th International Conference on Embedded Software and Systems (HPCC-ICESS), 2012 IEEE 14th International Conference on. IEEE, 2012.
|
||||
- Li, Y., Katsipoulakis, N.R., Chandramouli, B., Goldstein, J. and Kossmann, D., 2017. Mison: a fast JSON parser for data analytics. Proceedings of the VLDB Endowment, 10(10), pp.1118-1129. http://www.vldb.org/pvldb/vol10/p1118-li.pdf
|
||||
- Cameron, Robert D., et al. "Parallel scanning with bitstream addition: An xml case study." European Conference on Parallel Processing. Springer, Berlin, Heidelberg, 2011.
|
||||
- Cameron, Robert D., Kenneth S. Herdy, and Dan Lin. "High performance XML parsing using parallel bit stream technology." Proceedings of the 2008 conference of the center for advanced studies on collaborative research: meeting of minds. ACM, 2008.
|
||||
- Shah, Bhavik, et al. "A data parallel algorithm for XML DOM parsing." International XML Database Symposium. Springer, Berlin, Heidelberg, 2009.
|
||||
- Cameron, Robert D., and Dan Lin. "Architectural support for SWAR text processing with parallel bit streams: the inductive doubling principle." ACM Sigplan Notices. Vol. 44. No. 3. ACM, 2009.
|
||||
- Amagasa, Toshiyuki, Mana Seino, and Hiroyuki Kitagawa. "Energy-Efficient XML Stream Processing through Element-Skipping Parsing." Database and Expert Systems Applications (DEXA), 2013 24th International Workshop on. IEEE, 2013.
|
||||
- Medforth, Nigel Woodland. "icXML: Accelerating Xerces-C 3.1. 1 using the Parabix Framework." (2013).
|
||||
- Zhang, Qiang Scott. Embedding Parallel Bit Stream Technology Into Expat. Diss. Simon Fraser University, 2010.
|
||||
- Cameron, Robert D., et al. "Fast Regular Expression Matching with Bit-parallel Data Streams."
|
||||
- Lin, Dan. Bits filter: a high-performance multiple string pattern matching algorithm for malware detection. Diss. School of Computing Science-Simon Fraser University, 2010.
|
||||
- Yang, Shiyang. Validation of XML Document Based on Parallel Bit Stream Technology. Diss. Applied Sciences: School of Computing Science, 2013.
|
||||
- N. Nakasato, "Implementation of a parallel tree method on a GPU", Journal of Computational Science, vol. 3, no. 3, pp. 132-141, 2012.
|
||||
|
||||
|
||||
[license]:LICENSE
|
||||
[license img]:https://img.shields.io/badge/License-Apache%202-blue.svg
|
||||
For compilers that do not support [C++17](https://en.wikipedia.org/wiki/C%2B%2B17), we bundle the string-view library which is published under the Boost license (http://www.boost.org/LICENSE_1_0.txt). Like the Apache license, the Boost license is a permissive license allowing commercial redistribution.
|
||||
|
||||
+69
@@ -0,0 +1,69 @@
|
||||
# 0.5
|
||||
|
||||
## Highlights
|
||||
|
||||
Performance
|
||||
* Faster and simpler UTF-8 validation with the lookup4 algorithm https://github.com/simdjson/simdjson/pull/993
|
||||
* We improved the performance of simdjson under Visual Studio by about 25%. Users will still get better performance with clang-cl (+30%) but the gap has been reduced. https://github.com/simdjson/simdjson/pull/1031
|
||||
|
||||
Code usability
|
||||
* In `parse_many`, when parsing streams of JSON documetns, we give to the users runtime control as to whether threads are used (via the parser.threaded attribute). https://github.com/simdjson/simdjson/issues/925
|
||||
* Prefixed public macros to avoid name clashes with other libraries. https://github.com/simdjson/simdjson/issues/1035
|
||||
* Better documentation regarding package managers (brew, MSYS2, conan, apt, vcpkg, FreeBSD package manager, etc.).
|
||||
* Better documentation regarding CMake usage.
|
||||
|
||||
Standards
|
||||
* We improved standard compliance with respect to both the JSON RFC 8259 and JSON Pointer RFC 6901. We added the at_pointer method to nodes for standard-compliant JSON Pointer queries. The legacy `at(std::string_view)` method remains but is deprecated since it is not standard-compliant as per RFC 6901.
|
||||
* We removed computed GOTOs without sacrificing performance thus improving the C++ standard compliance (since computed GOTOs are compiler-specific extensions).
|
||||
* Better support for C++20 https://github.com/simdjson/simdjson/pull/1050
|
||||
|
||||
# 0.4
|
||||
|
||||
## Highlights
|
||||
|
||||
- Test coverage has been greatly improved and we have resolved many static-analysis warnings on different systems.
|
||||
- We added a fast (8GB/s) minifier that works directly on JSON strings.
|
||||
- We added fast (10GB/s) UTF-8 validator that works directly on strings (any strings, including non-JSON).
|
||||
- The array and object elements have a constant-time size() method.
|
||||
- Performance improvements to the API (type(), get<>()).
|
||||
- The parse_many function (ndjson) has been entirely reworked. It now uses a single secondary thread instead of several new threads.
|
||||
- We have introduced a faster UTF-8 validation algorithm (lookup3) for all kernels (ARM, x64 SSE, x64 AVX).
|
||||
- C++11 support for older compilers and systems.
|
||||
- FreeBSD support (and tests).
|
||||
- We support the clang front-end compiler (clangcl) under Visual Studio.
|
||||
- It is now possible to target ARM platforms under Visual Studio.
|
||||
- The simdjson library will never abort or print to standard output/error.
|
||||
|
||||
# 0.3
|
||||
|
||||
## Highlights
|
||||
|
||||
- **Multi-Document Parsing:** Read a bundle of JSON documents (ndjson) 2-4x faster than doing it
|
||||
individually. [API docs](https://github.com/simdjson/simdjson/blob/master/doc/basics.md#newline-delimited-json-ndjson-and-json-lines) / [Design Details](https://github.com/simdjson/simdjson/blob/master/doc/parse_many.md)
|
||||
- **Simplified API:** The API has been completely revamped for ease of use, including a new JSON
|
||||
navigation API and fluent support for error code *and* exception styles of error handling with a
|
||||
single API. [Docs](https://github.com/simdjson/simdjson/blob/master/doc/basics.md#the-basics-loading-and-parsing-json-documents)
|
||||
- **Exact Float Parsing:** Now simdjson parses floats flawlessly *without* any performance loss,
|
||||
thanks to [great work by @michaeleisel and @lemire](https://github.com/simdjson/simdjson/pull/558).
|
||||
[Blog Post](https://lemire.me/blog/2020/03/10/fast-float-parsing-in-practice/)
|
||||
- **Even Faster:** The fastest parser got faster! With a [shiny new UTF-8 validator](https://github.com/simdjson/simdjson/pull/387)
|
||||
and meticulously refactored SIMD core, simdjson 0.3 is 15% faster than before, running at 2.5 GB/s
|
||||
(where 0.2 ran at 2.2 GB/s).
|
||||
|
||||
## Minor Highlights
|
||||
|
||||
- Fallback implementation: simdjson now has a non-SIMD fallback implementation, and can run even on
|
||||
very old 64-bit machines.
|
||||
- Automatic allocation: as part of API simplification, the parser no longer has to be preallocated--
|
||||
it will adjust automatically when it encounters larger files.
|
||||
- Runtime selection API: We've exposed simdjson's runtime CPU detection and implementation selection
|
||||
as an API, so you can tell what implementation we detected and test with other implementations.
|
||||
- Error handling your way: Whether you use exceptions or check error codes, simdjson lets you handle
|
||||
errors in your style. APIs that can fail return simdjson_result<T>, letting you check the error
|
||||
code before using the result. But if you are more comfortable with exceptions, skip the error code
|
||||
and cast straight to T, and exceptions will be thrown automatically if an error happens. Use the
|
||||
same API either way!
|
||||
- Error chaining: We also worked to keep non-exception error-handling short and sweet. Instead of
|
||||
having to check the error code after every single operation, now you can *chain* JSON navigation
|
||||
calls like looking up an object field or array element, or casting to a string, so that you only
|
||||
have to check the error code once at the very end.
|
||||
-138
@@ -1,138 +0,0 @@
|
||||
#!/bin/bash
|
||||
########################################################################
|
||||
# Generates an "amalgamation build" for roaring. Inspired by similar
|
||||
# script used by whefs.
|
||||
########################################################################
|
||||
SCRIPTPATH="$( cd "$(dirname "$0")" ; pwd -P )"
|
||||
|
||||
echo "We are about to amalgamate all simdjson files into one source file. "
|
||||
echo "See https://www.sqlite.org/amalgamation.html and https://en.wikipedia.org/wiki/Single_Compilation_Unit for rationale. "
|
||||
|
||||
AMAL_H="simdjson.h"
|
||||
AMAL_C="simdjson.cpp"
|
||||
|
||||
# order does not matter
|
||||
ALLCFILES="
|
||||
$SCRIPTPATH/src/jsonioutil.cpp
|
||||
$SCRIPTPATH/src/jsonminifier.cpp
|
||||
$SCRIPTPATH/src/jsonparser.cpp
|
||||
$SCRIPTPATH/src/stage1_find_marks.cpp
|
||||
$SCRIPTPATH/src/stage2_build_tape.cpp
|
||||
$SCRIPTPATH/src/parsedjson.cpp
|
||||
$SCRIPTPATH/src/parsedjsoniterator.cpp
|
||||
"
|
||||
|
||||
# order matters
|
||||
ALLCHEADERS="
|
||||
$SCRIPTPATH/include/simdjson/simdjson_version.h
|
||||
$SCRIPTPATH/include/simdjson/portability.h
|
||||
$SCRIPTPATH/include/simdjson/common_defs.h
|
||||
$SCRIPTPATH/include/simdjson/jsoncharutils.h
|
||||
$SCRIPTPATH/include/simdjson/jsonformatutils.h
|
||||
$SCRIPTPATH/include/simdjson/jsonioutil.h
|
||||
$SCRIPTPATH/include/simdjson/simdprune_tables.h
|
||||
$SCRIPTPATH/include/simdjson/simdutf8check.h
|
||||
$SCRIPTPATH/include/simdjson/jsonminifier.h
|
||||
$SCRIPTPATH/include/simdjson/parsedjson.h
|
||||
$SCRIPTPATH/include/simdjson/stage1_find_marks.h
|
||||
$SCRIPTPATH/include/simdjson/stringparsing.h
|
||||
$SCRIPTPATH/include/simdjson/numberparsing.h
|
||||
$SCRIPTPATH/include/simdjson/stage2_build_tape.h
|
||||
$SCRIPTPATH/include/simdjson/jsonparser.h
|
||||
"
|
||||
|
||||
for i in ${ALLCHEADERS} ${ALLCFILES}; do
|
||||
test -e $i && continue
|
||||
echo "FATAL: source file [$i] not found."
|
||||
exit 127
|
||||
done
|
||||
|
||||
|
||||
function stripinc()
|
||||
{
|
||||
sed -e '/# *include *"/d' -e '/# *include *<simdjson\//d'
|
||||
}
|
||||
function dofile()
|
||||
{
|
||||
echo "/* begin file $1 */"
|
||||
# echo "#line 8 \"$1\"" ## redefining the line/file is not nearly as useful as it sounds for debugging. It breaks IDEs.
|
||||
stripinc < $1
|
||||
echo "/* end file $1 */"
|
||||
}
|
||||
|
||||
timestamp=$(date)
|
||||
echo "Creating ${AMAL_H}..."
|
||||
echo "/* auto-generated on ${timestamp}. Do not edit! */" > "${AMAL_H}"
|
||||
{
|
||||
for h in ${ALLCHEADERS}; do
|
||||
dofile $h
|
||||
done
|
||||
} >> "${AMAL_H}"
|
||||
|
||||
|
||||
echo "Creating ${AMAL_C}..."
|
||||
echo "/* auto-generated on ${timestamp}. Do not edit! */" > "${AMAL_C}"
|
||||
{
|
||||
echo "#include \"${AMAL_H}\""
|
||||
|
||||
echo ""
|
||||
echo "/* used for http://dmalloc.com/ Dmalloc - Debug Malloc Library */"
|
||||
echo "#ifdef DMALLOC"
|
||||
echo "#include \"dmalloc.h\""
|
||||
echo "#endif"
|
||||
echo ""
|
||||
|
||||
for h in ${ALLCFILES}; do
|
||||
dofile $h
|
||||
done
|
||||
} >> "${AMAL_C}"
|
||||
|
||||
|
||||
|
||||
DEMOCPP="amalgamation_demo.cpp"
|
||||
echo "Creating ${DEMOCPP}..."
|
||||
echo "/* auto-generated on ${timestamp}. Do not edit! */" > "${DEMOCPP}"
|
||||
cat <<< '
|
||||
#include <iostream>
|
||||
#include "simdjson.h"
|
||||
#include "simdjson.cpp"
|
||||
int main(int argc, char *argv[]) {
|
||||
const char * filename = argv[1];
|
||||
std::string_view p = get_corpus(filename);
|
||||
ParsedJson pj = build_parsed_json(p); // do the parsing
|
||||
if( ! pj.isValid() ) {
|
||||
std::cout << "not valid" << std::endl;
|
||||
} else {
|
||||
std::cout << "valid" << std::endl;
|
||||
}
|
||||
return EXIT_SUCCESS;
|
||||
}
|
||||
' >> "${DEMOCPP}"
|
||||
|
||||
echo "Done with all files generation. "
|
||||
|
||||
echo "Files have been written to directory: $PWD "
|
||||
ls -la ${AMAL_C} ${AMAL_H} ${DEMOCPP}
|
||||
|
||||
echo "Giving final instructions:"
|
||||
|
||||
|
||||
CPPBIN=${DEMOCPP%%.*}
|
||||
|
||||
echo "Try :"
|
||||
echo "c++ -march=native -O3 -std=c++17 -o ${CPPBIN} ${DEMOCPP} && ./${CPPBIN} ../jsonexamples/twitter.json "
|
||||
|
||||
SINGLEHDR=$SCRIPTPATH/singleheader
|
||||
echo "Copying files to $SCRIPTPATH/singleheader "
|
||||
mkdir -p $SINGLEHDR
|
||||
echo "c++ -march=native -O3 -std=c++17 -o ${CPPBIN} ${DEMOCPP} && ./${CPPBIN} ../jsonexamples/twitter.json " > $SINGLEHDR/README.md
|
||||
cp ${AMAL_C} ${AMAL_H} ${DEMOCPP} $SINGLEHDR
|
||||
ls $SINGLEHDR
|
||||
|
||||
cd $SINGLEHDR && c++ -march=native -O3 -std=c++17 -o ${CPPBIN} ${DEMOCPP} && ./${CPPBIN} ../jsonexamples/twitter.json
|
||||
|
||||
lowercase(){
|
||||
echo "$1" | tr 'A-Z' 'a-z'
|
||||
}
|
||||
|
||||
OS=`lowercase \`uname\``
|
||||
@@ -1,8 +1,50 @@
|
||||
target_include_directories(${SIMDJSON_LIB_NAME}
|
||||
PUBLIC
|
||||
${PROJECT_SOURCE_DIR}/benchmark
|
||||
${PROJECT_SOURCE_DIR}/benchmark/linux
|
||||
)
|
||||
include_directories( . linux )
|
||||
link_libraries(simdjson-windows-headers test-data)
|
||||
|
||||
add_cpp_benchmark(parse)
|
||||
add_cpp_benchmark(statisticalmodel)
|
||||
|
||||
if (TARGET benchmark::benchmark)
|
||||
add_executable(bench_sax bench_sax.cpp)
|
||||
target_link_libraries(bench_sax PRIVATE simdjson-internal-flags simdjson-include-source benchmark::benchmark)
|
||||
endif (TARGET benchmark::benchmark)
|
||||
|
||||
link_libraries(simdjson simdjson-flags)
|
||||
add_executable(benchfeatures benchfeatures.cpp)
|
||||
add_executable(get_corpus_benchmark get_corpus_benchmark.cpp)
|
||||
add_executable(perfdiff perfdiff.cpp)
|
||||
add_executable(parse parse.cpp)
|
||||
add_executable(parse_stream parse_stream.cpp)
|
||||
add_executable(statisticalmodel statisticalmodel.cpp)
|
||||
|
||||
add_executable(parse_noutf8validation parse.cpp)
|
||||
target_compile_definitions(parse_noutf8validation PRIVATE SIMDJSON_SKIPUTF8VALIDATION)
|
||||
add_executable(parse_nonumberparsing parse.cpp)
|
||||
target_compile_definitions(parse_nonumberparsing PRIVATE SIMDJSON_SKIPNUMBERPARSING)
|
||||
add_executable(parse_nostringparsing parse.cpp)
|
||||
target_compile_definitions(parse_nostringparsing PRIVATE SIMDJSON_SKIPSTRINGPARSING)
|
||||
|
||||
if (TARGET competition-all)
|
||||
add_executable(distinctuseridcompetition distinctuseridcompetition.cpp)
|
||||
target_link_libraries(distinctuseridcompetition PRIVATE competition-core)
|
||||
|
||||
add_executable(minifiercompetition minifiercompetition.cpp)
|
||||
target_link_libraries(minifiercompetition PRIVATE competition-core)
|
||||
|
||||
add_executable(parseandstatcompetition parseandstatcompetition.cpp)
|
||||
target_link_libraries(parseandstatcompetition PRIVATE competition-core)
|
||||
|
||||
add_executable(parsingcompetition parsingcompetition.cpp)
|
||||
target_link_libraries(parsingcompetition PRIVATE competition-core)
|
||||
|
||||
add_executable(allparsingcompetition parsingcompetition.cpp)
|
||||
target_link_libraries(allparsingcompetition PRIVATE competition-all)
|
||||
target_compile_definitions(allparsingcompetition PRIVATE ALLPARSER)
|
||||
endif()
|
||||
|
||||
if (TARGET benchmark::benchmark)
|
||||
link_libraries(benchmark::benchmark)
|
||||
add_executable(bench_parse_call bench_parse_call.cpp)
|
||||
add_executable(bench_dom_api bench_dom_api.cpp)
|
||||
add_executable(bench_ondemand bench_ondemand.cpp)
|
||||
endif()
|
||||
|
||||
include(checkperf.cmake)
|
||||
|
||||
@@ -0,0 +1,19 @@
|
||||
# From the ROOT, run:
|
||||
# docker build -t simdjsonbench -f benchmark/Dockerfile . && docker run --privileged -t simdjsonbench
|
||||
FROM gcc:8.3
|
||||
|
||||
# # Build latest
|
||||
# ENV latest_release=v0.2.1
|
||||
# WORKDIR /usr/src/$latest_release/
|
||||
# RUN git clone --depth 1 https://github.com/lemire/simdjson/ -b $latest_release .
|
||||
# RUN make parse
|
||||
|
||||
# # Build master
|
||||
# WORKDIR /usr/src/master/
|
||||
# RUN git clone --depth 1 https://github.com/lemire/simdjson/ .
|
||||
# RUN make parse
|
||||
|
||||
# Build the current source
|
||||
COPY . /usr/src/current/
|
||||
WORKDIR /usr/src/current/
|
||||
RUN make checkperf
|
||||
@@ -0,0 +1,745 @@
|
||||
#include <benchmark/benchmark.h>
|
||||
#include "simdjson.h"
|
||||
#include <sstream>
|
||||
|
||||
using namespace simdjson;
|
||||
using namespace benchmark;
|
||||
using namespace std;
|
||||
|
||||
const padded_string EMPTY_ARRAY("[]", 2);
|
||||
|
||||
static const char *TWITTER_JSON = SIMDJSON_BENCHMARK_DATA_DIR "twitter.json";
|
||||
static const char *NUMBERS_JSON = SIMDJSON_BENCHMARK_DATA_DIR "numbers.json";
|
||||
|
||||
static void recover_one_string(State& state) {
|
||||
dom::parser parser;
|
||||
const std::string_view data = "\"one string\"";
|
||||
padded_string docdata{data};
|
||||
// we do not want mem. alloc. in the loop.
|
||||
auto error = parser.allocate(docdata.size());
|
||||
if(error) {
|
||||
cout << error << endl;
|
||||
return;
|
||||
}
|
||||
dom::element doc;
|
||||
if ((error = parser.parse(docdata).get(doc))) {
|
||||
cerr << "could not parse string" << error << endl;
|
||||
return;
|
||||
}
|
||||
for (simdjson_unused auto _ : state) {
|
||||
std::string_view v;
|
||||
error = doc.get(v);
|
||||
if (error) {
|
||||
cerr << "could not get string" << error << endl;
|
||||
return;
|
||||
}
|
||||
benchmark::DoNotOptimize(v);
|
||||
}
|
||||
}
|
||||
BENCHMARK(recover_one_string);
|
||||
|
||||
|
||||
static void serialize_twitter(State& state) {
|
||||
dom::parser parser;
|
||||
padded_string docdata;
|
||||
auto error = padded_string::load(TWITTER_JSON).get(docdata);
|
||||
if(error) {
|
||||
cerr << "could not parse twitter.json" << error << endl;
|
||||
return;
|
||||
}
|
||||
// we do not want mem. alloc. in the loop.
|
||||
if((error = parser.allocate(docdata.size()))) {
|
||||
cout << error << endl;
|
||||
return;
|
||||
}
|
||||
dom::element doc;
|
||||
if ((error = parser.parse(docdata).get(doc))) {
|
||||
cerr << "could not parse twitter.json" << error << endl;
|
||||
return;
|
||||
}
|
||||
size_t bytes = 0;
|
||||
for (simdjson_unused auto _ : state) {
|
||||
std::string serial = simdjson::minify(doc);
|
||||
bytes += serial.size();
|
||||
benchmark::DoNotOptimize(serial);
|
||||
}
|
||||
// we validate the result
|
||||
{
|
||||
auto serial = simdjson::minify(doc);
|
||||
dom::element doc2; // we parse the minified output
|
||||
if ((error = parser.parse(serial).get(doc2))) { throw std::runtime_error("serialization error"); }
|
||||
auto serial2 = simdjson::minify(doc2); // we minify a second time
|
||||
if(serial != serial2) { throw std::runtime_error("serialization mismatch"); }
|
||||
}
|
||||
// Gigabyte: https://en.wikipedia.org/wiki/Gigabyte
|
||||
state.counters["Gigabytes"] = benchmark::Counter(
|
||||
double(bytes), benchmark::Counter::kIsRate,
|
||||
benchmark::Counter::OneK::kIs1000); // For GiB : kIs1024
|
||||
state.counters["docs"] = Counter(double(state.iterations()), benchmark::Counter::kIsRate);
|
||||
}
|
||||
BENCHMARK(serialize_twitter)->ComputeStatistics("max", [](const std::vector<double>& v) -> double {
|
||||
return *(std::max_element(std::begin(v), std::end(v)));
|
||||
})->DisplayAggregatesOnly(true);
|
||||
|
||||
|
||||
static void serialize_big_string_to_string(State& state) {
|
||||
dom::parser parser;
|
||||
std::vector<char> content;
|
||||
content.push_back('\"');
|
||||
for(size_t i = 0 ; i < 100000; i ++) {
|
||||
content.push_back('0' + char(i%10)); // we add what looks like a long list of digits
|
||||
}
|
||||
content.push_back('\"');
|
||||
dom::element doc;
|
||||
simdjson::error_code error;
|
||||
if ((error = parser.parse(content.data(), content.size()).get(doc))) {
|
||||
cerr << "could not parse big string" << error << endl;
|
||||
return;
|
||||
}
|
||||
size_t bytes = 0;
|
||||
for (simdjson_unused auto _ : state) {
|
||||
auto serial = simdjson::to_string(doc);
|
||||
bytes += serial.size();
|
||||
benchmark::DoNotOptimize(serial);
|
||||
}
|
||||
// Gigabyte: https://en.wikipedia.org/wiki/Gigabyte
|
||||
state.counters["Gigabytes"] = benchmark::Counter(
|
||||
double(bytes), benchmark::Counter::kIsRate,
|
||||
benchmark::Counter::OneK::kIs1000); // For GiB : kIs1024
|
||||
state.counters["docs"] = Counter(double(state.iterations()), benchmark::Counter::kIsRate);
|
||||
}
|
||||
BENCHMARK(serialize_big_string_to_string)->ComputeStatistics("max", [](const std::vector<double>& v) -> double {
|
||||
return *(std::max_element(std::begin(v), std::end(v)));
|
||||
})->DisplayAggregatesOnly(true);
|
||||
|
||||
|
||||
static void serialize_twitter_to_string(State& state) {
|
||||
dom::parser parser;
|
||||
padded_string docdata;
|
||||
auto error = padded_string::load(TWITTER_JSON).get(docdata);
|
||||
if(error) {
|
||||
cerr << "could not parse twitter.json" << error << endl;
|
||||
return;
|
||||
}
|
||||
// we do not want mem. alloc. in the loop.
|
||||
if((error = parser.allocate(docdata.size()))) {
|
||||
cout << error << endl;
|
||||
return;
|
||||
}
|
||||
dom::element doc;
|
||||
if ((error = parser.parse(docdata).get(doc))) {
|
||||
cerr << "could not parse twitter.json" << error << endl;
|
||||
return;
|
||||
}
|
||||
size_t bytes = 0;
|
||||
for (simdjson_unused auto _ : state) {
|
||||
auto serial = simdjson::to_string(doc);
|
||||
bytes += serial.size();
|
||||
benchmark::DoNotOptimize(serial);
|
||||
}
|
||||
// we validate the result
|
||||
{
|
||||
auto serial = simdjson::to_string(doc);
|
||||
dom::element doc2; // we parse the stringify output
|
||||
if ((error = parser.parse(serial).get(doc2))) { throw std::runtime_error("serialization error"); }
|
||||
auto serial2 = simdjson::to_string(doc2); // we stringify again
|
||||
if(serial != serial2) { throw std::runtime_error("serialization mismatch"); }
|
||||
}
|
||||
// Gigabyte: https://en.wikipedia.org/wiki/Gigabyte
|
||||
state.counters["Gigabytes"] = benchmark::Counter(
|
||||
double(bytes), benchmark::Counter::kIsRate,
|
||||
benchmark::Counter::OneK::kIs1000); // For GiB : kIs1024
|
||||
state.counters["docs"] = Counter(double(state.iterations()), benchmark::Counter::kIsRate);
|
||||
}
|
||||
BENCHMARK(serialize_twitter_to_string)->ComputeStatistics("max", [](const std::vector<double>& v) -> double {
|
||||
return *(std::max_element(std::begin(v), std::end(v)));
|
||||
})->DisplayAggregatesOnly(true);
|
||||
|
||||
static void serialize_twitter_string_builder(State& state) {
|
||||
dom::parser parser;
|
||||
padded_string docdata;
|
||||
auto error = padded_string::load(TWITTER_JSON).get(docdata);
|
||||
if(error) {
|
||||
cerr << "could not parse twitter.json" << error << endl;
|
||||
return;
|
||||
}
|
||||
// we do not want mem. alloc. in the loop.
|
||||
if((error = parser.allocate(docdata.size()))) {
|
||||
cout << error << endl;
|
||||
return;
|
||||
}
|
||||
dom::element doc;
|
||||
if ((error = parser.parse(docdata).get(doc))) {
|
||||
cerr << "could not parse twitter.json" << error << endl;
|
||||
return;
|
||||
}
|
||||
size_t bytes = 0;
|
||||
simdjson::internal::string_builder<> sb;// not part of our public API, for internal use
|
||||
for (simdjson_unused auto _ : state) {
|
||||
sb.clear();
|
||||
sb.append(doc);
|
||||
std::string_view serial = sb.str();
|
||||
bytes += serial.size();
|
||||
benchmark::DoNotOptimize(serial);
|
||||
}
|
||||
// Gigabyte: https://en.wikipedia.org/wiki/Gigabyte
|
||||
state.counters["Gigabytes"] = benchmark::Counter(
|
||||
double(bytes), benchmark::Counter::kIsRate,
|
||||
benchmark::Counter::OneK::kIs1000); // For GiB : kIs1024
|
||||
state.counters["docs"] = Counter(double(state.iterations()), benchmark::Counter::kIsRate);
|
||||
}
|
||||
BENCHMARK(serialize_twitter_string_builder)->ComputeStatistics("max", [](const std::vector<double>& v) -> double {
|
||||
return *(std::max_element(std::begin(v), std::end(v)));
|
||||
})->DisplayAggregatesOnly(true);
|
||||
|
||||
|
||||
static void numbers_scan(State& state) {
|
||||
// Prints the number of results in twitter.json
|
||||
dom::parser parser;
|
||||
dom::array arr;
|
||||
simdjson::error_code error;
|
||||
if ((error = parser.load(NUMBERS_JSON).get(arr))) {
|
||||
cerr << "could not read " << NUMBERS_JSON << " as an array: " << error << endl;
|
||||
return;
|
||||
}
|
||||
for (simdjson_unused auto _ : state) {
|
||||
std::vector<double> container;
|
||||
for (auto e : arr) {
|
||||
double x;
|
||||
if ((error = e.get(x))) { cerr << "found a node that is not an number: " << error << endl; break;}
|
||||
container.push_back(x);
|
||||
}
|
||||
benchmark::DoNotOptimize(container.data());
|
||||
benchmark::ClobberMemory();
|
||||
}
|
||||
}
|
||||
BENCHMARK(numbers_scan);
|
||||
|
||||
static void numbers_size_scan(State& state) {
|
||||
// Prints the number of results in twitter.json
|
||||
dom::parser parser;
|
||||
dom::array arr;
|
||||
simdjson::error_code error;
|
||||
if ((error = parser.load(NUMBERS_JSON).get(arr))) {
|
||||
cerr << "could not read " << NUMBERS_JSON << " as an array: " << error << endl;
|
||||
return;
|
||||
}
|
||||
for (simdjson_unused auto _ : state) {
|
||||
std::vector<double> container;
|
||||
container.resize(arr.size());
|
||||
size_t pos = 0;
|
||||
for (auto e : arr) {
|
||||
double x;
|
||||
if ((error = e.get(x))) { cerr << "found a node that is not an number: " << error << endl; break;}
|
||||
container[pos++] = x;
|
||||
}
|
||||
if(pos != container.size()) { cerr << "bad count" << endl; }
|
||||
benchmark::DoNotOptimize(container.data());
|
||||
benchmark::ClobberMemory();
|
||||
}
|
||||
}
|
||||
BENCHMARK(numbers_size_scan);
|
||||
|
||||
|
||||
static void numbers_type_scan(State& state) {
|
||||
// Prints the number of results in twitter.json
|
||||
dom::parser parser;
|
||||
dom::array arr;
|
||||
simdjson::error_code error;
|
||||
if ((error = parser.load(NUMBERS_JSON).get(arr))) {
|
||||
cerr << "could not read " << NUMBERS_JSON << " as an array" << endl;
|
||||
return;
|
||||
}
|
||||
for (simdjson_unused auto _ : state) {
|
||||
std::vector<double> container;
|
||||
for (auto e : arr) {
|
||||
dom::element_type actual_type = e.type();
|
||||
if(actual_type != dom::element_type::DOUBLE) {
|
||||
cerr << "found a node that is not an number?" << endl; break;
|
||||
}
|
||||
double x;
|
||||
error = e.get(x);
|
||||
container.push_back(x);
|
||||
}
|
||||
benchmark::DoNotOptimize(container.data());
|
||||
benchmark::ClobberMemory();
|
||||
}
|
||||
}
|
||||
BENCHMARK(numbers_type_scan);
|
||||
|
||||
static void numbers_type_size_scan(State& state) {
|
||||
// Prints the number of results in twitter.json
|
||||
dom::parser parser;
|
||||
dom::array arr;
|
||||
simdjson::error_code error;
|
||||
if ((error = parser.load(NUMBERS_JSON).get(arr))) {
|
||||
cerr << "could not read " << NUMBERS_JSON << " as an array: " << error << endl;
|
||||
return;
|
||||
}
|
||||
for (simdjson_unused auto _ : state) {
|
||||
std::vector<double> container;
|
||||
container.resize(arr.size());
|
||||
size_t pos = 0;
|
||||
for (auto e : arr) {
|
||||
dom::element_type actual_type = e.type();
|
||||
if(actual_type != dom::element_type::DOUBLE) {
|
||||
cerr << "found a node that is not an number?" << endl; break;
|
||||
}
|
||||
double x;
|
||||
error = e.get(x);
|
||||
container[pos++] = x;
|
||||
}
|
||||
if(pos != container.size()) { cerr << "bad count" << endl; }
|
||||
benchmark::DoNotOptimize(container.data());
|
||||
benchmark::ClobberMemory();
|
||||
}
|
||||
}
|
||||
BENCHMARK(numbers_type_size_scan);
|
||||
|
||||
static void numbers_load_scan(State& state) {
|
||||
// Prints the number of results in twitter.json
|
||||
dom::parser parser;
|
||||
dom::array arr;
|
||||
simdjson::error_code error;
|
||||
for (simdjson_unused auto _ : state) {
|
||||
// this may hit the disk, but probably just once
|
||||
if ((error = parser.load(NUMBERS_JSON).get(arr))) {
|
||||
cerr << "could not read " << NUMBERS_JSON << " as an array: " << error << endl;
|
||||
break;
|
||||
}
|
||||
std::vector<double> container;
|
||||
for (auto e : arr) {
|
||||
double x;
|
||||
if ((error = e.get(x))) { cerr << "found a node that is not an number: " << error << endl; break;}
|
||||
container.push_back(x);
|
||||
}
|
||||
benchmark::DoNotOptimize(container.data());
|
||||
benchmark::ClobberMemory();
|
||||
}
|
||||
}
|
||||
BENCHMARK(numbers_load_scan);
|
||||
|
||||
static void numbers_load_size_scan(State& state) {
|
||||
// Prints the number of results in twitter.json
|
||||
dom::parser parser;
|
||||
dom::array arr;
|
||||
simdjson::error_code error;
|
||||
for (simdjson_unused auto _ : state) {
|
||||
// this may hit the disk, but probably just once
|
||||
if ((error = parser.load(NUMBERS_JSON).get(arr))) {
|
||||
cerr << "could not read " << NUMBERS_JSON << " as an array" << endl;
|
||||
break;
|
||||
}
|
||||
std::vector<double> container;
|
||||
container.resize(arr.size());
|
||||
size_t pos = 0;
|
||||
for (auto e : arr) {
|
||||
double x;
|
||||
if ((error = e.get(x))) { cerr << "found a node that is not an number?" << endl; break;}
|
||||
container[pos++] = x;
|
||||
}
|
||||
if(pos != container.size()) { cerr << "bad count" << endl; }
|
||||
benchmark::DoNotOptimize(container.data());
|
||||
benchmark::ClobberMemory();
|
||||
}
|
||||
}
|
||||
BENCHMARK(numbers_load_size_scan);
|
||||
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
|
||||
|
||||
static void numbers_exceptions_scan(State& state) {
|
||||
// Prints the number of results in twitter.json
|
||||
dom::parser parser;
|
||||
dom::array arr = parser.load(NUMBERS_JSON);
|
||||
for (simdjson_unused auto _ : state) {
|
||||
std::vector<double> container;
|
||||
for (double x : arr) {
|
||||
container.push_back(x);
|
||||
}
|
||||
benchmark::DoNotOptimize(container.data());
|
||||
benchmark::ClobberMemory();
|
||||
}
|
||||
}
|
||||
BENCHMARK(numbers_exceptions_scan);
|
||||
|
||||
static void numbers_exceptions_size_scan(State& state) {
|
||||
// Prints the number of results in twitter.json
|
||||
dom::parser parser;
|
||||
dom::array arr = parser.load(NUMBERS_JSON);
|
||||
for (simdjson_unused auto _ : state) {
|
||||
std::vector<double> container;
|
||||
container.resize(arr.size());
|
||||
size_t pos = 0;
|
||||
for (auto e : arr) {
|
||||
container[pos++] = double(e);
|
||||
}
|
||||
if(pos != container.size()) { cerr << "bad count" << endl; }
|
||||
benchmark::DoNotOptimize(container.data());
|
||||
benchmark::ClobberMemory();
|
||||
}
|
||||
}
|
||||
BENCHMARK(numbers_exceptions_size_scan);
|
||||
|
||||
|
||||
|
||||
static void numbers_type_exceptions_scan(State& state) {
|
||||
// Prints the number of results in twitter.json
|
||||
dom::parser parser;
|
||||
dom::array arr = parser.load(NUMBERS_JSON);
|
||||
for (simdjson_unused auto _ : state) {
|
||||
std::vector<double> container;
|
||||
for (auto e : arr) {
|
||||
dom::element_type actual_type = e.type();
|
||||
if(actual_type != dom::element_type::DOUBLE) {
|
||||
cerr << "found a node that is not an number?" << endl; break;
|
||||
}
|
||||
container.push_back(double(e));
|
||||
}
|
||||
benchmark::DoNotOptimize(container.data());
|
||||
benchmark::ClobberMemory();
|
||||
}
|
||||
}
|
||||
BENCHMARK(numbers_type_exceptions_scan);
|
||||
|
||||
static void numbers_type_exceptions_size_scan(State& state) {
|
||||
// Prints the number of results in twitter.json
|
||||
dom::parser parser;
|
||||
dom::array arr = parser.load(NUMBERS_JSON);
|
||||
for (simdjson_unused auto _ : state) {
|
||||
std::vector<double> container;
|
||||
container.resize(arr.size());
|
||||
size_t pos = 0;
|
||||
for (auto e : arr) {
|
||||
dom::element_type actual_type = e.type();
|
||||
if(actual_type != dom::element_type::DOUBLE) {
|
||||
cerr << "found a node that is not an number?" << endl; break;
|
||||
}
|
||||
container[pos++] = double(e);
|
||||
}
|
||||
if(pos != container.size()) { cerr << "bad count" << endl; }
|
||||
benchmark::DoNotOptimize(container.data());
|
||||
benchmark::ClobberMemory();
|
||||
}
|
||||
}
|
||||
BENCHMARK(numbers_type_exceptions_size_scan);
|
||||
|
||||
static void numbers_exceptions_load_scan(State& state) {
|
||||
// Prints the number of results in twitter.json
|
||||
dom::parser parser;
|
||||
for (simdjson_unused auto _ : state) {
|
||||
// this may hit the disk, but probably just once
|
||||
dom::array arr = parser.load(NUMBERS_JSON);
|
||||
std::vector<double> container;
|
||||
for (double x : arr) {
|
||||
container.push_back(x);
|
||||
}
|
||||
benchmark::DoNotOptimize(container.data());
|
||||
benchmark::ClobberMemory();
|
||||
}
|
||||
}
|
||||
BENCHMARK(numbers_exceptions_load_scan);
|
||||
|
||||
static void numbers_exceptions_load_size_scan(State& state) {
|
||||
// Prints the number of results in twitter.json
|
||||
dom::parser parser;
|
||||
for (simdjson_unused auto _ : state) {
|
||||
// this may hit the disk, but probably just once
|
||||
dom::array arr = parser.load(NUMBERS_JSON);
|
||||
std::vector<double> container;
|
||||
container.resize(arr.size());
|
||||
size_t pos = 0;
|
||||
for (double x : arr) {
|
||||
container[pos++] = x;
|
||||
}
|
||||
if(pos != container.size()) { cerr << "bad count" << endl; }
|
||||
benchmark::DoNotOptimize(container.data());
|
||||
benchmark::ClobberMemory();
|
||||
}
|
||||
}
|
||||
BENCHMARK(numbers_exceptions_load_size_scan);
|
||||
|
||||
|
||||
static void twitter_count(State& state) {
|
||||
// Prints the number of results in twitter.json
|
||||
dom::parser parser;
|
||||
dom::element doc = parser.load(TWITTER_JSON);
|
||||
for (simdjson_unused auto _ : state) {
|
||||
uint64_t result_count = doc["search_metadata"]["count"];
|
||||
if (result_count != 100) { return; }
|
||||
}
|
||||
}
|
||||
BENCHMARK(twitter_count);
|
||||
|
||||
#ifndef SIMDJSON_DISABLE_DEPRECATED_API
|
||||
SIMDJSON_PUSH_DISABLE_WARNINGS
|
||||
SIMDJSON_DISABLE_DEPRECATED_WARNING
|
||||
static void iterator_twitter_count(State& state) {
|
||||
// Prints the number of results in twitter.json
|
||||
padded_string json = padded_string::load(TWITTER_JSON);
|
||||
ParsedJson pj = build_parsed_json(json);
|
||||
for (simdjson_unused auto _ : state) {
|
||||
ParsedJson::Iterator iter(pj);
|
||||
// uint64_t result_count = doc["search_metadata"]["count"];
|
||||
if (!iter.move_to_key("search_metadata")) { return; }
|
||||
if (!iter.move_to_key("count")) { return; }
|
||||
if (!iter.is_integer()) { return; }
|
||||
int64_t result_count = iter.get_integer();
|
||||
|
||||
if (result_count != 100) { return; }
|
||||
}
|
||||
}
|
||||
BENCHMARK(iterator_twitter_count);
|
||||
SIMDJSON_POP_DISABLE_WARNINGS
|
||||
#endif // SIMDJSON_DISABLE_DEPRECATED_API
|
||||
|
||||
static void twitter_default_profile(State& state) {
|
||||
// Count unique users with a default profile.
|
||||
dom::parser parser;
|
||||
dom::element doc = parser.load(TWITTER_JSON);
|
||||
for (simdjson_unused auto _ : state) {
|
||||
set<string_view> default_users;
|
||||
for (dom::object tweet : doc["statuses"]) {
|
||||
dom::object user = tweet["user"];
|
||||
if (user["default_profile"]) {
|
||||
default_users.insert(user["screen_name"]);
|
||||
}
|
||||
}
|
||||
if (default_users.size() != 86) { return; }
|
||||
}
|
||||
}
|
||||
BENCHMARK(twitter_default_profile);
|
||||
|
||||
static void twitter_image_sizes(State& state) {
|
||||
// Count unique image sizes
|
||||
dom::parser parser;
|
||||
dom::element doc = parser.load(TWITTER_JSON);
|
||||
simdjson::error_code error;
|
||||
for (simdjson_unused auto _ : state) {
|
||||
set<tuple<uint64_t, uint64_t>> image_sizes;
|
||||
for (dom::object tweet : doc["statuses"]) {
|
||||
dom::array media;
|
||||
if (not (error = tweet["entities"]["media"].get(media))) {
|
||||
for (dom::object image : media) {
|
||||
for (auto size : image["sizes"].get<dom::object>()) {
|
||||
image_sizes.insert({ size.value["w"], size.value["h"] });
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if (image_sizes.size() != 15) { return; };
|
||||
}
|
||||
}
|
||||
BENCHMARK(twitter_image_sizes);
|
||||
|
||||
#endif // SIMDJSON_EXCEPTIONS
|
||||
|
||||
static void error_code_twitter_count(State& state) noexcept {
|
||||
// Prints the number of results in twitter.json
|
||||
dom::parser parser;
|
||||
simdjson::error_code error;
|
||||
dom::element doc;
|
||||
if ((error = parser.load(TWITTER_JSON).get(doc))) { return; }
|
||||
for (simdjson_unused auto _ : state) {
|
||||
uint64_t value;
|
||||
if ((error = doc["search_metadata"]["count"].get(value))) { return; }
|
||||
if (value != 100) { return; }
|
||||
}
|
||||
}
|
||||
BENCHMARK(error_code_twitter_count);
|
||||
|
||||
static void error_code_twitter_default_profile(State& state) noexcept {
|
||||
// Count unique users with a default profile.
|
||||
dom::parser parser;
|
||||
simdjson::error_code error;
|
||||
dom::element doc;
|
||||
if ((error = parser.load(TWITTER_JSON).get(doc))) { std::cerr << error << std::endl; return; }
|
||||
for (simdjson_unused auto _ : state) {
|
||||
set<string_view> default_users;
|
||||
|
||||
dom::array tweets;
|
||||
if ((error = doc["statuses"].get(tweets))) { return; }
|
||||
for (dom::element tweet : tweets) {
|
||||
dom::object user;
|
||||
if ((error = tweet["user"].get(user))) { return; }
|
||||
bool default_profile;
|
||||
if ((error = user["default_profile"].get(default_profile))) { return; }
|
||||
if (default_profile) {
|
||||
std::string_view screen_name;
|
||||
if ((error = user["screen_name"].get(screen_name))) { return; }
|
||||
default_users.insert(screen_name);
|
||||
}
|
||||
}
|
||||
|
||||
if (default_users.size() != 86) { return; }
|
||||
}
|
||||
}
|
||||
BENCHMARK(error_code_twitter_default_profile);
|
||||
|
||||
#ifndef SIMDJSON_DISABLE_DEPRECATED_API
|
||||
|
||||
SIMDJSON_PUSH_DISABLE_WARNINGS
|
||||
SIMDJSON_DISABLE_DEPRECATED_WARNING
|
||||
|
||||
static void iterator_twitter_default_profile(State& state) {
|
||||
// Count unique users with a default profile.
|
||||
padded_string json;
|
||||
auto error = padded_string::load(TWITTER_JSON).get(json);
|
||||
if (error) { std::cerr << error << std::endl; return; }
|
||||
ParsedJson pj = build_parsed_json(json);
|
||||
for (simdjson_unused auto _ : state) {
|
||||
set<string_view> default_users;
|
||||
ParsedJson::Iterator iter(pj);
|
||||
|
||||
// for (dom::object tweet : doc["statuses"]) {
|
||||
if (!(iter.move_to_key("statuses") && iter.is_array())) { return; }
|
||||
if (iter.down()) { // first status
|
||||
do {
|
||||
|
||||
// dom::object user = tweet["user"];
|
||||
if (!(iter.move_to_key("user") && iter.is_object())) { return; }
|
||||
|
||||
// if (user["default_profile"]) {
|
||||
if (iter.move_to_key("default_profile")) {
|
||||
if (iter.is_true()) {
|
||||
if (!iter.up()) { return; } // back to user
|
||||
|
||||
// default_users.insert(user["screen_name"]);
|
||||
if (!(iter.move_to_key("screen_name") && iter.is_string())) { return; }
|
||||
default_users.insert(string_view(iter.get_string(), iter.get_string_length()));
|
||||
}
|
||||
if (!iter.up()) { return; } // back to user
|
||||
}
|
||||
|
||||
if (!iter.up()) { return; } // back to status
|
||||
|
||||
} while (iter.next()); // next status
|
||||
}
|
||||
|
||||
if (default_users.size() != 86) { return; }
|
||||
}
|
||||
}
|
||||
|
||||
SIMDJSON_POP_DISABLE_WARNINGS
|
||||
BENCHMARK(iterator_twitter_default_profile);
|
||||
#endif // SIMDJSON_DISABLE_DEPRECATED_API
|
||||
|
||||
static void error_code_twitter_image_sizes(State& state) noexcept {
|
||||
// Count unique image sizes
|
||||
dom::parser parser;
|
||||
simdjson::error_code error;
|
||||
dom::element doc;
|
||||
if ((error = parser.load(TWITTER_JSON).get(doc))) { std::cerr << error << std::endl; return; }
|
||||
for (simdjson_unused auto _ : state) {
|
||||
set<tuple<uint64_t, uint64_t>> image_sizes;
|
||||
dom::array statuses;
|
||||
if ((error = doc["statuses"].get(statuses))) { return; }
|
||||
for (dom::element tweet : statuses) {
|
||||
dom::array images;
|
||||
if (not (error = tweet["entities"]["media"].get(images))) {
|
||||
for (dom::element image : images) {
|
||||
dom::object sizes;
|
||||
if ((error = image["sizes"].get(sizes))) { return; }
|
||||
for (auto size : sizes) {
|
||||
uint64_t width, height;
|
||||
if ((error = size.value["w"].get(width))) { return; }
|
||||
if ((error = size.value["h"].get(height))) { return; }
|
||||
image_sizes.insert({ width, height });
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if (image_sizes.size() != 15) { return; };
|
||||
}
|
||||
}
|
||||
BENCHMARK(error_code_twitter_image_sizes);
|
||||
|
||||
#ifndef SIMDJSON_DISABLE_DEPRECATED_API
|
||||
|
||||
SIMDJSON_PUSH_DISABLE_WARNINGS
|
||||
SIMDJSON_DISABLE_DEPRECATED_WARNING
|
||||
static void iterator_twitter_image_sizes(State& state) {
|
||||
// Count unique image sizes
|
||||
padded_string json;
|
||||
auto error = padded_string::load(TWITTER_JSON).get(json);
|
||||
if (error) { std::cerr << error << std::endl; return; }
|
||||
ParsedJson pj = build_parsed_json(json);
|
||||
for (simdjson_unused auto _ : state) {
|
||||
set<tuple<uint64_t, uint64_t>> image_sizes;
|
||||
ParsedJson::Iterator iter(pj);
|
||||
|
||||
// for (dom::object tweet : doc["statuses"]) {
|
||||
if (!(iter.move_to_key("statuses") && iter.is_array())) { return; }
|
||||
if (iter.down()) { // first status
|
||||
do {
|
||||
|
||||
// dom::object media;
|
||||
// not_found = tweet["entities"]["media"].get(media);
|
||||
// if (!not_found) {
|
||||
if (iter.move_to_key("entities")) {
|
||||
if (!iter.is_object()) { return; }
|
||||
if (iter.move_to_key("media")) {
|
||||
if (!iter.is_array()) { return; }
|
||||
|
||||
// for (dom::object image : media) {
|
||||
if (iter.down()) { // first media
|
||||
do {
|
||||
|
||||
// for (auto [key, size] : dom::object(image["sizes"])) {
|
||||
if (!(iter.move_to_key("sizes") && iter.is_object())) { return; }
|
||||
if (iter.down()) { // first size
|
||||
do {
|
||||
iter.move_to_value();
|
||||
|
||||
// image_sizes.insert({ size["w"], size["h"] });
|
||||
if (!(iter.move_to_key("w")) && !iter.is_integer()) { return; }
|
||||
uint64_t width = iter.get_integer();
|
||||
if (!iter.up()) { return; } // back to size
|
||||
if (!(iter.move_to_key("h")) && !iter.is_integer()) { return; }
|
||||
uint64_t height = iter.get_integer();
|
||||
if (!iter.up()) { return; } // back to size
|
||||
image_sizes.insert({ width, height });
|
||||
|
||||
} while (iter.next()); // next size
|
||||
if (!iter.up()) { return; } // back to sizes
|
||||
}
|
||||
if (!iter.up()) { return; } // back to image
|
||||
} while (iter.next()); // next image
|
||||
if (!iter.up()) { return; } // back to media
|
||||
}
|
||||
if (!iter.up()) { return; } // back to entities
|
||||
}
|
||||
if (!iter.up()) { return; } // back to status
|
||||
}
|
||||
} while (iter.next()); // next status
|
||||
}
|
||||
|
||||
if (image_sizes.size() != 15) { return; };
|
||||
}
|
||||
}
|
||||
BENCHMARK(iterator_twitter_image_sizes);
|
||||
|
||||
#endif // SIMDJSON_DISABLE_DEPRECATED_API
|
||||
|
||||
#ifndef SIMDJSON_DISABLE_DEPRECATED_API
|
||||
static void print_json(State& state) noexcept {
|
||||
// Prints the number of results in twitter.json
|
||||
dom::parser parser;
|
||||
|
||||
padded_string json;
|
||||
auto error = padded_string::load(TWITTER_JSON).get(json);
|
||||
if (error) { std::cerr << error << std::endl; return; }
|
||||
|
||||
int code = json_parse(json, parser);
|
||||
if (code) { cerr << error_message(code) << endl; return; }
|
||||
for (simdjson_unused auto _ : state) {
|
||||
std::stringstream s;
|
||||
if (!parser.print_json(s)) { cerr << "print_json failed" << endl; return; }
|
||||
}
|
||||
}
|
||||
BENCHMARK(print_json);
|
||||
#endif // SIMDJSON_DISABLE_DEPRECATED_API
|
||||
SIMDJSON_POP_DISABLE_WARNINGS
|
||||
|
||||
BENCHMARK_MAIN();
|
||||
@@ -0,0 +1,26 @@
|
||||
#include "simdjson.h"
|
||||
#include <iostream>
|
||||
#include <sstream>
|
||||
#include <random>
|
||||
#include <vector>
|
||||
SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
|
||||
#include <benchmark/benchmark.h>
|
||||
SIMDJSON_POP_DISABLE_WARNINGS
|
||||
|
||||
#include "partial_tweets/ondemand.h"
|
||||
#include "partial_tweets/iter.h"
|
||||
#include "partial_tweets/dom.h"
|
||||
|
||||
#include "largerandom/ondemand.h"
|
||||
#include "largerandom/iter.h"
|
||||
#include "largerandom/dom.h"
|
||||
|
||||
#include "kostya/ondemand.h"
|
||||
#include "kostya/iter.h"
|
||||
#include "kostya/dom.h"
|
||||
|
||||
#include "distinctuserid/ondemand.h"
|
||||
#include "distinctuserid/dom.h"
|
||||
|
||||
|
||||
BENCHMARK_MAIN();
|
||||
@@ -0,0 +1,197 @@
|
||||
#include <benchmark/benchmark.h>
|
||||
#include "simdjson.h"
|
||||
using namespace simdjson;
|
||||
using namespace benchmark;
|
||||
using namespace std;
|
||||
|
||||
const padded_string EMPTY_ARRAY("[]", 2);
|
||||
const char *TWITTER_JSON = SIMDJSON_BENCHMARK_DATA_DIR "twitter.json";
|
||||
const char *GSOC_JSON = SIMDJSON_BENCHMARK_DATA_DIR "gsoc-2018.json";
|
||||
|
||||
|
||||
|
||||
static void unicode_validate_twitter(State& state) {
|
||||
dom::parser parser;
|
||||
padded_string docdata;
|
||||
auto error = padded_string::load(TWITTER_JSON).get(docdata);
|
||||
if(error) {
|
||||
cerr << "could not parse twitter.json" << error << endl;
|
||||
return;
|
||||
}
|
||||
// we do not want mem. alloc. in the loop.
|
||||
error = parser.allocate(docdata.size());
|
||||
if(error) {
|
||||
cout << error << endl;
|
||||
return;
|
||||
}
|
||||
size_t bytes = 0;
|
||||
for (simdjson_unused auto _ : state) {
|
||||
bool is_ok = simdjson::validate_utf8(docdata.data(), docdata.size());
|
||||
bytes += docdata.size();
|
||||
benchmark::DoNotOptimize(is_ok);
|
||||
}
|
||||
// Gigabyte: https://en.wikipedia.org/wiki/Gigabyte
|
||||
state.counters["Gigabytes"] = benchmark::Counter(
|
||||
double(bytes), benchmark::Counter::kIsRate,
|
||||
benchmark::Counter::OneK::kIs1000); // For GiB : kIs1024
|
||||
state.counters["docs"] = Counter(double(state.iterations()), benchmark::Counter::kIsRate);
|
||||
}
|
||||
BENCHMARK(unicode_validate_twitter)->Repetitions(10)->ComputeStatistics("max", [](const std::vector<double>& v) -> double {
|
||||
return *(std::max_element(std::begin(v), std::end(v)));
|
||||
})->DisplayAggregatesOnly(true);
|
||||
|
||||
static void parse_twitter(State& state) {
|
||||
dom::parser parser;
|
||||
padded_string docdata;
|
||||
auto error = padded_string::load(TWITTER_JSON).get(docdata);
|
||||
if(error) {
|
||||
cerr << "could not parse twitter.json" << error << endl;
|
||||
return;
|
||||
}
|
||||
// we do not want mem. alloc. in the loop.
|
||||
error = parser.allocate(docdata.size());
|
||||
if(error) {
|
||||
cout << error << endl;
|
||||
return;
|
||||
}
|
||||
size_t bytes = 0;
|
||||
for (simdjson_unused auto _ : state) {
|
||||
dom::element doc;
|
||||
bytes += docdata.size();
|
||||
if ((error = parser.parse(docdata).get(doc))) {
|
||||
cerr << "could not parse twitter.json" << error << endl;
|
||||
return;
|
||||
}
|
||||
benchmark::DoNotOptimize(doc);
|
||||
}
|
||||
// Gigabyte: https://en.wikipedia.org/wiki/Gigabyte
|
||||
state.counters["Gigabytes"] = benchmark::Counter(
|
||||
double(bytes), benchmark::Counter::kIsRate,
|
||||
benchmark::Counter::OneK::kIs1000); // For GiB : kIs1024
|
||||
state.counters["docs"] = Counter(double(state.iterations()), benchmark::Counter::kIsRate);
|
||||
}
|
||||
BENCHMARK(parse_twitter)->Repetitions(10)->ComputeStatistics("max", [](const std::vector<double>& v) -> double {
|
||||
return *(std::max_element(std::begin(v), std::end(v)));
|
||||
})->DisplayAggregatesOnly(true);
|
||||
|
||||
|
||||
static void parse_gsoc(State& state) {
|
||||
dom::parser parser;
|
||||
padded_string docdata;
|
||||
auto error = padded_string::load(GSOC_JSON).get(docdata);
|
||||
if(error) {
|
||||
cerr << "could not parse gsoc-2018.json" << error << endl;
|
||||
return;
|
||||
}
|
||||
// we do not want mem. alloc. in the loop.
|
||||
error = parser.allocate(docdata.size());
|
||||
if(error) {
|
||||
cout << error << endl;
|
||||
return;
|
||||
}
|
||||
size_t bytes = 0;
|
||||
for (simdjson_unused auto _ : state) {
|
||||
bytes += docdata.size();
|
||||
dom::element doc;
|
||||
if ((error = parser.parse(docdata).get(doc))) {
|
||||
cerr << "could not parse gsoc-2018.json" << error << endl;
|
||||
return;
|
||||
}
|
||||
benchmark::DoNotOptimize(doc);
|
||||
}
|
||||
// Gigabyte: https://en.wikipedia.org/wiki/Gigabyte
|
||||
state.counters["Gigabytes"] = benchmark::Counter(
|
||||
double(bytes), benchmark::Counter::kIsRate,
|
||||
benchmark::Counter::OneK::kIs1000); // For GiB : kIs1024
|
||||
state.counters["docs"] = Counter(double(state.iterations()), benchmark::Counter::kIsRate);
|
||||
}
|
||||
BENCHMARK(parse_gsoc)->Repetitions(10)->ComputeStatistics("max", [](const std::vector<double>& v) -> double {
|
||||
return *(std::max_element(std::begin(v), std::end(v)));
|
||||
})->DisplayAggregatesOnly(true);
|
||||
|
||||
|
||||
|
||||
#ifndef SIMDJSON_DISABLE_DEPRECATED_API
|
||||
SIMDJSON_PUSH_DISABLE_WARNINGS
|
||||
SIMDJSON_DISABLE_DEPRECATED_WARNING
|
||||
static void json_parse(State& state) {
|
||||
ParsedJson pj;
|
||||
if (!pj.allocate_capacity(EMPTY_ARRAY.length())) { return; }
|
||||
for (simdjson_unused auto _ : state) {
|
||||
auto error = json_parse(EMPTY_ARRAY, pj);
|
||||
if (error) { return; }
|
||||
}
|
||||
}
|
||||
SIMDJSON_POP_DISABLE_WARNINGS
|
||||
BENCHMARK(json_parse);
|
||||
#endif // SIMDJSON_DISABLE_DEPRECATED_API
|
||||
|
||||
static void parser_parse_error_code(State& state) {
|
||||
dom::parser parser;
|
||||
if (parser.allocate(EMPTY_ARRAY.length())) { return; }
|
||||
for (simdjson_unused auto _ : state) {
|
||||
auto error = parser.parse(EMPTY_ARRAY).error();
|
||||
if (error) { return; }
|
||||
}
|
||||
}
|
||||
BENCHMARK(parser_parse_error_code);
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
|
||||
static void parser_parse_exception(State& state) {
|
||||
dom::parser parser;
|
||||
if (parser.allocate(EMPTY_ARRAY.length())) { return; }
|
||||
for (simdjson_unused auto _ : state) {
|
||||
try {
|
||||
simdjson_unused dom::element doc = parser.parse(EMPTY_ARRAY);
|
||||
} catch(simdjson_error &j) {
|
||||
cout << j.what() << endl;
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
BENCHMARK(parser_parse_exception);
|
||||
|
||||
#endif // SIMDJSON_EXCEPTIONS
|
||||
|
||||
#ifndef SIMDJSON_DISABLE_DEPRECATED_API
|
||||
SIMDJSON_PUSH_DISABLE_WARNINGS
|
||||
SIMDJSON_DISABLE_DEPRECATED_WARNING
|
||||
static void build_parsed_json(State& state) {
|
||||
for (simdjson_unused auto _ : state) {
|
||||
dom::parser parser = simdjson::build_parsed_json(EMPTY_ARRAY);
|
||||
if (!parser.valid) { return; }
|
||||
}
|
||||
}
|
||||
SIMDJSON_POP_DISABLE_WARNINGS
|
||||
|
||||
BENCHMARK(build_parsed_json);
|
||||
#endif
|
||||
|
||||
static void document_parse_error_code(State& state) {
|
||||
for (simdjson_unused auto _ : state) {
|
||||
dom::parser parser;
|
||||
auto error = parser.parse(EMPTY_ARRAY).error();
|
||||
if (error) { return; }
|
||||
}
|
||||
}
|
||||
BENCHMARK(document_parse_error_code);
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
|
||||
static void document_parse_exception(State& state) {
|
||||
for (simdjson_unused auto _ : state) {
|
||||
try {
|
||||
dom::parser parser;
|
||||
simdjson_unused dom::element doc = parser.parse(EMPTY_ARRAY);
|
||||
} catch(simdjson_error &j) {
|
||||
cout << j.what() << endl;
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
BENCHMARK(document_parse_exception);
|
||||
|
||||
#endif // SIMDJSON_EXCEPTIONS
|
||||
|
||||
BENCHMARK_MAIN();
|
||||
@@ -0,0 +1,14 @@
|
||||
#include "simdjson.h"
|
||||
#include "simdjson.cpp"
|
||||
#include <iostream>
|
||||
#include <sstream>
|
||||
#include <random>
|
||||
#include <vector>
|
||||
SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
|
||||
#include <benchmark/benchmark.h>
|
||||
SIMDJSON_POP_DISABLE_WARNINGS
|
||||
|
||||
#include "partial_tweets/sax.h"
|
||||
#include "largerandom/sax.h"
|
||||
|
||||
BENCHMARK_MAIN();
|
||||
@@ -0,0 +1,472 @@
|
||||
#include "event_counter.h"
|
||||
|
||||
#include <cassert>
|
||||
#include <cctype>
|
||||
#ifndef _MSC_VER
|
||||
#include <dirent.h>
|
||||
#endif
|
||||
#include <unistd.h>
|
||||
#include <cinttypes>
|
||||
#include <initializer_list>
|
||||
|
||||
#include <cstdio>
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
|
||||
#include <algorithm>
|
||||
#include <chrono>
|
||||
#include <cstring>
|
||||
#include <fstream>
|
||||
#include <iomanip>
|
||||
#include <iostream>
|
||||
#include <map>
|
||||
#include <set>
|
||||
#include <sstream>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "linux-perf-events.h"
|
||||
#ifdef __linux__
|
||||
#include <libgen.h>
|
||||
#endif
|
||||
|
||||
#include "simdjson.h"
|
||||
|
||||
#include <functional>
|
||||
|
||||
#include "benchmarker.h"
|
||||
|
||||
using namespace simdjson;
|
||||
using std::cerr;
|
||||
using std::cout;
|
||||
using std::endl;
|
||||
using std::string;
|
||||
using std::to_string;
|
||||
using std::vector;
|
||||
using std::ostream;
|
||||
using std::ofstream;
|
||||
using std::exception;
|
||||
|
||||
// Stash the exe_name in main() for functions to use
|
||||
char* exe_name;
|
||||
|
||||
void print_usage(ostream& out) {
|
||||
out << "Usage: " << exe_name << " [-v] [-n #] [-s STAGE] [-a ARCH]" << endl;
|
||||
out << endl;
|
||||
out << "Runs the parser against jsonexamples/generated json files in a loop, measuring speed and other statistics." << endl;
|
||||
out << endl;
|
||||
out << "Options:" << endl;
|
||||
out << endl;
|
||||
out << "-n # - Number of iterations per file. Default: 400" << endl;
|
||||
out << "-i # - Number of times to iterate a single file before moving to the next. Default: 20" << endl;
|
||||
out << "-v - Verbose output." << endl;
|
||||
out << "-s STAGE - Stop after the given stage." << endl;
|
||||
out << " -s stage1 - Stop after find_structural_bits." << endl;
|
||||
out << " -s all - Run all stages." << endl;
|
||||
out << "-a ARCH - Use the parser with the designated architecture (HASWELL, WESTMERE," << endl;
|
||||
out << " PPC64 or ARM64). By default, detects best supported architecture." << endl;
|
||||
}
|
||||
|
||||
void exit_usage(string message) {
|
||||
cerr << message << endl;
|
||||
cerr << endl;
|
||||
print_usage(cerr);
|
||||
exit(EXIT_FAILURE);
|
||||
}
|
||||
|
||||
struct option_struct {
|
||||
bool stage1_only = false;
|
||||
|
||||
int32_t iterations = 400;
|
||||
int32_t iteration_step = 50;
|
||||
|
||||
bool verbose = false;
|
||||
|
||||
option_struct(int argc, char **argv) {
|
||||
int c;
|
||||
|
||||
while ((c = getopt(argc, argv, "vtn:i:a:s:")) != -1) {
|
||||
switch (c) {
|
||||
case 'n':
|
||||
iterations = atoi(optarg);
|
||||
break;
|
||||
case 'i':
|
||||
iteration_step = atoi(optarg);
|
||||
break;
|
||||
case 'v':
|
||||
verbose = true;
|
||||
break;
|
||||
case 'a': {
|
||||
auto impl = simdjson::available_implementations[optarg];
|
||||
if(impl && impl->supported_by_runtime_system()) {
|
||||
simdjson::active_implementation = impl;
|
||||
} else {
|
||||
std::cerr << "implementation " << optarg << " not found or not supported " << std::endl;
|
||||
}
|
||||
}
|
||||
break;
|
||||
case 's':
|
||||
if (!strcmp(optarg, "stage1")) {
|
||||
stage1_only = true;
|
||||
} else if (!strcmp(optarg, "all")) {
|
||||
stage1_only = false;
|
||||
} else {
|
||||
exit_usage(string("Unsupported option value -s ") + optarg + ": expected -s stage1 or all");
|
||||
}
|
||||
break;
|
||||
default:
|
||||
exit_error(string("Unexpected argument ") + std::string(1,static_cast<char>(c)));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template<typename F>
|
||||
void each_stage(const F& f) const {
|
||||
f(BenchmarkStage::STAGE1);
|
||||
if (!this->stage1_only) {
|
||||
f(BenchmarkStage::STAGE2);
|
||||
f(BenchmarkStage::ALL);
|
||||
}
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
struct feature_benchmarker {
|
||||
benchmarker utf8;
|
||||
benchmarker utf8_miss;
|
||||
benchmarker escape;
|
||||
benchmarker escape_miss;
|
||||
benchmarker empty;
|
||||
benchmarker empty_miss;
|
||||
benchmarker struct7;
|
||||
benchmarker struct7_miss;
|
||||
benchmarker struct7_full;
|
||||
benchmarker struct15;
|
||||
benchmarker struct15_miss;
|
||||
benchmarker struct23;
|
||||
benchmarker struct23_miss;
|
||||
|
||||
feature_benchmarker(event_collector& collector) :
|
||||
utf8 (SIMDJSON_BENCHMARK_DATA_DIR "generated/utf-8.json", collector),
|
||||
utf8_miss (SIMDJSON_BENCHMARK_DATA_DIR "generated/utf-8-miss.json", collector),
|
||||
escape (SIMDJSON_BENCHMARK_DATA_DIR "generated/escape.json", collector),
|
||||
escape_miss (SIMDJSON_BENCHMARK_DATA_DIR "generated/escape-miss.json", collector),
|
||||
empty (SIMDJSON_BENCHMARK_DATA_DIR "generated/0-structurals.json", collector),
|
||||
empty_miss (SIMDJSON_BENCHMARK_DATA_DIR "generated/0-structurals-miss.json", collector),
|
||||
struct7 (SIMDJSON_BENCHMARK_DATA_DIR "generated/7-structurals.json", collector),
|
||||
struct7_miss (SIMDJSON_BENCHMARK_DATA_DIR "generated/7-structurals-miss.json", collector),
|
||||
struct7_full (SIMDJSON_BENCHMARK_DATA_DIR "generated/7-structurals-full.json", collector),
|
||||
struct15 (SIMDJSON_BENCHMARK_DATA_DIR "generated/15-structurals.json", collector),
|
||||
struct15_miss(SIMDJSON_BENCHMARK_DATA_DIR "generated/15-structurals-miss.json", collector),
|
||||
struct23 (SIMDJSON_BENCHMARK_DATA_DIR "generated/23-structurals.json", collector),
|
||||
struct23_miss(SIMDJSON_BENCHMARK_DATA_DIR "generated/23-structurals-miss.json", collector)
|
||||
{
|
||||
|
||||
}
|
||||
|
||||
simdjson_really_inline void run_iterations(size_t iterations, bool stage1_only=false) {
|
||||
struct7.run_iterations(iterations, stage1_only);
|
||||
struct7_miss.run_iterations(iterations, stage1_only);
|
||||
struct7_full.run_iterations(iterations, stage1_only);
|
||||
utf8.run_iterations(iterations, stage1_only);
|
||||
utf8_miss.run_iterations(iterations, stage1_only);
|
||||
escape.run_iterations(iterations, stage1_only);
|
||||
escape_miss.run_iterations(iterations, stage1_only);
|
||||
empty.run_iterations(iterations, stage1_only);
|
||||
empty_miss.run_iterations(iterations, stage1_only);
|
||||
struct15.run_iterations(iterations, stage1_only);
|
||||
struct15_miss.run_iterations(iterations, stage1_only);
|
||||
struct23.run_iterations(iterations, stage1_only);
|
||||
struct23_miss.run_iterations(iterations, stage1_only);
|
||||
}
|
||||
|
||||
double cost_per_block(BenchmarkStage stage, const benchmarker& feature, size_t feature_blocks, const benchmarker& base) const {
|
||||
return (feature[stage].best.elapsed_ns() - base[stage].best.elapsed_ns()) / double(feature_blocks);
|
||||
}
|
||||
|
||||
// Whether we're recording cache miss and branch miss events
|
||||
bool has_events() const {
|
||||
return empty.collector.has_events();
|
||||
}
|
||||
|
||||
// Base cost of any block (including empty ones)
|
||||
double base_cost(BenchmarkStage stage) const {
|
||||
return (empty[stage].best.elapsed_ns() / double(empty.stats->blocks));
|
||||
}
|
||||
|
||||
// Extra cost of a 1-7 structural block over an empty block
|
||||
double struct1_7_cost(BenchmarkStage stage) const {
|
||||
return cost_per_block(stage, struct7, struct7.stats->blocks_with_1_structural, empty);
|
||||
}
|
||||
// Extra cost of an 1-7-structural miss
|
||||
double struct1_7_miss_cost(BenchmarkStage stage) const {
|
||||
return cost_per_block(stage, struct7_miss, struct7_miss.stats->blocks_with_1_structural, struct7);
|
||||
}
|
||||
// Rate of 1-7-structural misses per 8-structural flip
|
||||
double struct1_7_miss_rate(BenchmarkStage stage) const {
|
||||
if (!has_events()) { return 1; }
|
||||
return struct7_miss[stage].best.branch_misses() - struct7[stage].best.branch_misses() / double(struct7_miss.stats->blocks_with_1_structural_flipped);
|
||||
}
|
||||
|
||||
// Extra cost of an 8-15 structural block over a 1-7 structural block
|
||||
double struct8_15_cost(BenchmarkStage stage) const {
|
||||
return cost_per_block(stage, struct15, struct15.stats->blocks_with_8_structurals, struct7);
|
||||
}
|
||||
// Extra cost of an 8-15-structural miss over a 1-7 miss
|
||||
double struct8_15_miss_cost(BenchmarkStage stage) const {
|
||||
return cost_per_block(stage, struct15_miss, struct15_miss.stats->blocks_with_8_structurals_flipped, struct15);
|
||||
}
|
||||
// Rate of 8-15-structural misses per 8-structural flip
|
||||
double struct8_15_miss_rate(BenchmarkStage stage) const {
|
||||
if (!has_events()) { return 1; }
|
||||
return double(struct15_miss[stage].best.branch_misses() - struct15[stage].best.branch_misses()) / double(struct15_miss.stats->blocks_with_8_structurals_flipped);
|
||||
}
|
||||
|
||||
// Extra cost of a 16+-structural block over an 8-15 structural block (actual varies based on # of structurals!)
|
||||
double struct16_cost(BenchmarkStage stage) const {
|
||||
return cost_per_block(stage, struct23, struct23.stats->blocks_with_16_structurals, struct15);
|
||||
}
|
||||
// Extra cost of a 16-structural miss over an 8-15 miss
|
||||
double struct16_miss_cost(BenchmarkStage stage) const {
|
||||
return cost_per_block(stage, struct23_miss, struct23_miss.stats->blocks_with_16_structurals_flipped, struct23);
|
||||
}
|
||||
// Rate of 16-structural misses per 16-structural flip
|
||||
double struct16_miss_rate(BenchmarkStage stage) const {
|
||||
if (!has_events()) { return 1; }
|
||||
return double(struct23_miss[stage].best.branch_misses() - struct23[stage].best.branch_misses()) / double(struct23_miss.stats->blocks_with_16_structurals_flipped);
|
||||
}
|
||||
|
||||
// Extra cost of having UTF-8 in a block
|
||||
double utf8_cost(BenchmarkStage stage) const {
|
||||
return cost_per_block(stage, utf8, utf8.stats->blocks_with_utf8, struct7_full);
|
||||
}
|
||||
// Extra cost of a UTF-8 miss
|
||||
double utf8_miss_cost(BenchmarkStage stage) const {
|
||||
return cost_per_block(stage, utf8_miss, utf8_miss.stats->blocks_with_utf8_flipped, utf8);
|
||||
}
|
||||
// Rate of UTF-8 misses per UTF-8 flip
|
||||
double utf8_miss_rate(BenchmarkStage stage) const {
|
||||
if (!has_events()) { return 1; }
|
||||
return double(utf8_miss[stage].best.branch_misses() - utf8[stage].best.branch_misses()) / double(utf8_miss.stats->blocks_with_utf8_flipped);
|
||||
}
|
||||
|
||||
// Extra cost of having escapes in a block
|
||||
double escape_cost(BenchmarkStage stage) const {
|
||||
return cost_per_block(stage, escape, escape.stats->blocks_with_escapes, struct7_full);
|
||||
}
|
||||
// Extra cost of an escape miss
|
||||
double escape_miss_cost(BenchmarkStage stage) const {
|
||||
return cost_per_block(stage, escape_miss, escape_miss.stats->blocks_with_escapes_flipped, escape);
|
||||
}
|
||||
// Rate of escape misses per escape flip
|
||||
double escape_miss_rate(BenchmarkStage stage) const {
|
||||
if (!has_events()) { return 1; }
|
||||
return double(escape_miss[stage].best.branch_misses() - escape[stage].best.branch_misses()) / double(escape_miss.stats->blocks_with_escapes_flipped);
|
||||
}
|
||||
|
||||
double calc_expected_feature_cost(BenchmarkStage stage, const benchmarker& file) const {
|
||||
// Expected base ns/block (empty)
|
||||
json_stats& stats = *file.stats;
|
||||
double expected = base_cost(stage) * double(stats.blocks);
|
||||
expected += struct1_7_cost(stage) * double(stats.blocks_with_1_structural);
|
||||
expected += utf8_cost(stage) * double(stats.blocks_with_utf8);
|
||||
expected += escape_cost(stage) * double(stats.blocks_with_escapes);
|
||||
expected += struct8_15_cost(stage) * double(stats.blocks_with_8_structurals);
|
||||
expected += struct16_cost(stage) * double(stats.blocks_with_16_structurals);
|
||||
return expected / double(stats.blocks);
|
||||
}
|
||||
|
||||
double calc_expected_miss_cost(BenchmarkStage stage, const benchmarker& file) const {
|
||||
// Expected base ns/block (empty)
|
||||
json_stats& stats = *file.stats;
|
||||
double expected = struct1_7_miss_cost(stage) * double(stats.blocks_with_1_structural_flipped) * struct1_7_miss_rate(stage);
|
||||
expected += utf8_miss_cost(stage) * double(stats.blocks_with_utf8_flipped) * utf8_miss_rate(stage);
|
||||
expected += escape_miss_cost(stage) * double(stats.blocks_with_escapes_flipped) * escape_miss_rate(stage);
|
||||
expected += struct8_15_miss_cost(stage) * double(stats.blocks_with_8_structurals_flipped) * struct8_15_miss_rate(stage);
|
||||
expected += struct16_miss_cost(stage) * double(stats.blocks_with_16_structurals_flipped) * struct16_miss_rate(stage);
|
||||
return expected / double(stats.blocks);
|
||||
}
|
||||
|
||||
double calc_expected_misses(BenchmarkStage stage, const benchmarker& file) const {
|
||||
json_stats& stats = *file.stats;
|
||||
double expected = double(stats.blocks_with_1_structural_flipped) * struct1_7_miss_rate(stage);
|
||||
expected += double(stats.blocks_with_utf8_flipped) * utf8_miss_rate(stage);
|
||||
expected += double(stats.blocks_with_escapes_flipped) * escape_miss_rate(stage);
|
||||
expected += double(stats.blocks_with_8_structurals_flipped) * struct8_15_miss_rate(stage);
|
||||
expected += double(stats.blocks_with_16_structurals_flipped) * struct16_miss_rate(stage);
|
||||
return expected;
|
||||
}
|
||||
|
||||
double calc_expected(BenchmarkStage stage, const benchmarker& file) const {
|
||||
return calc_expected_feature_cost(stage, file) + calc_expected_miss_cost(stage, file);
|
||||
}
|
||||
|
||||
void print(const option_struct& options) const {
|
||||
printf("\n");
|
||||
printf("Features in ns/block (64 bytes):\n");
|
||||
printf("\n");
|
||||
printf("| %-8s ", "Stage");
|
||||
printf("| %8s ", "Base");
|
||||
printf("| %8s ", "7 Struct");
|
||||
printf("| %8s ", "UTF-8");
|
||||
printf("| %8s ", "Escape");
|
||||
printf("| %8s ", "15 Str.");
|
||||
printf("| %8s ", "16+ Str.");
|
||||
printf("| %15s ", "7 Struct Miss");
|
||||
printf("| %15s ", "UTF-8 Miss");
|
||||
printf("| %15s ", "Escape Miss");
|
||||
printf("| %15s ", "15 Str. Miss");
|
||||
printf("| %15s ", "16+ Str. Miss");
|
||||
printf("|\n");
|
||||
|
||||
printf("|%.10s", "---------------------------------------");
|
||||
printf("|%.10s", "---------------------------------------");
|
||||
printf("|%.10s", "---------------------------------------");
|
||||
printf("|%.10s", "---------------------------------------");
|
||||
printf("|%.10s", "---------------------------------------");
|
||||
printf("|%.10s", "---------------------------------------");
|
||||
printf("|%.10s", "---------------------------------------");
|
||||
printf("|%.17s", "---------------------------------------");
|
||||
printf("|%.17s", "---------------------------------------");
|
||||
printf("|%.17s", "---------------------------------------");
|
||||
printf("|%.17s", "---------------------------------------");
|
||||
printf("|%.17s", "---------------------------------------");
|
||||
printf("|\n");
|
||||
|
||||
options.each_stage([&](auto stage) {
|
||||
printf("| %-8s ", benchmark_stage_name(stage));
|
||||
printf("| %8.3g ", base_cost(stage));
|
||||
printf("| %8.3g ", struct1_7_cost(stage));
|
||||
printf("| %8.3g ", utf8_cost(stage));
|
||||
printf("| %8.3g ", escape_cost(stage));
|
||||
printf("| %8.3g ", struct8_15_cost(stage));
|
||||
printf("| %8.3g ", struct16_cost(stage));
|
||||
if (has_events()) {
|
||||
printf("| %8.3g (%3d%%) ", struct1_7_miss_cost(stage), int(struct1_7_miss_rate(stage)*100));
|
||||
printf("| %8.3g (%3d%%) ", utf8_miss_cost(stage), int(utf8_miss_rate(stage)*100));
|
||||
printf("| %8.3g (%3d%%) ", escape_miss_cost(stage), int(escape_miss_rate(stage)*100));
|
||||
printf("| %8.3g (%3d%%) ", struct8_15_miss_cost(stage), int(struct8_15_miss_rate(stage)*100));
|
||||
printf("| %8.3g (%3d%%) ", struct16_miss_cost(stage), int(struct16_miss_rate(stage)*100));
|
||||
} else {
|
||||
printf("| %8.3g ", struct1_7_miss_cost(stage));
|
||||
printf("| %8.3g ", utf8_miss_cost(stage));
|
||||
printf("| %8.3g ", escape_miss_cost(stage));
|
||||
printf("| %8.3g ", struct8_15_miss_cost(stage));
|
||||
printf("| %8.3g ", struct16_miss_cost(stage));
|
||||
}
|
||||
printf("|\n");
|
||||
});
|
||||
}
|
||||
};
|
||||
|
||||
void print_file_effectiveness(BenchmarkStage stage, const char* filename, const benchmarker& results, const feature_benchmarker& features) {
|
||||
double actual = results[stage].best.elapsed_ns() / double(results.stats->blocks);
|
||||
double calc = features.calc_expected(stage, results);
|
||||
double actual_misses = results[stage].best.branch_misses();
|
||||
double calc_misses = features.calc_expected_misses(stage, results);
|
||||
double calc_miss_cost = features.calc_expected_miss_cost(stage, results);
|
||||
printf(" | %-8s ", benchmark_stage_name(stage));
|
||||
printf("| %-15s ", filename);
|
||||
printf("| %8.3g ", features.calc_expected_feature_cost(stage, results));
|
||||
printf("| %8.3g ", calc_miss_cost);
|
||||
printf("| %8.3g ", calc);
|
||||
printf("| %8.3g ", actual);
|
||||
printf("| %+8.3g ", actual - calc);
|
||||
printf("| %13llu ", (long long unsigned)(calc_misses));
|
||||
if (features.has_events()) {
|
||||
printf("| %13llu ", (long long unsigned)(actual_misses));
|
||||
printf("| %+13lld ", (long long int)(actual_misses - calc_misses));
|
||||
double miss_adjustment = calc_miss_cost * (double(int64_t(actual_misses - calc_misses)) / calc_misses);
|
||||
printf("| %8.3g ", calc_miss_cost + miss_adjustment);
|
||||
printf("| %+8.3g ", actual - (calc + miss_adjustment));
|
||||
}
|
||||
printf("|\n");
|
||||
}
|
||||
|
||||
int main(int argc, char *argv[]) {
|
||||
// Read options
|
||||
exe_name = argv[0];
|
||||
option_struct options(argc, argv);
|
||||
if (options.verbose) {
|
||||
verbose_stream = &cout;
|
||||
}
|
||||
|
||||
// Initialize the event collector. We put this early so if it prints an error message, it's the
|
||||
// first thing printed.
|
||||
event_collector collector;
|
||||
|
||||
// Set up benchmarkers by reading all files
|
||||
feature_benchmarker features(collector);
|
||||
benchmarker gsoc_2018(SIMDJSON_BENCHMARK_DATA_DIR "gsoc-2018.json", collector);
|
||||
benchmarker twitter(SIMDJSON_BENCHMARK_DATA_DIR "twitter.json", collector);
|
||||
benchmarker random(SIMDJSON_BENCHMARK_DATA_DIR "random.json", collector);
|
||||
|
||||
// Run the benchmarks
|
||||
progress_bar progress(options.iterations, 100);
|
||||
// Put the if (options.stage1_only) *outside* the loop so that run_iterations will be optimized
|
||||
if (options.stage1_only) {
|
||||
for (int iteration = 0; iteration < options.iterations; iteration += options.iteration_step) {
|
||||
if (!options.verbose) { progress.print(iteration); }
|
||||
features.run_iterations(options.iteration_step, true);
|
||||
gsoc_2018.run_iterations(options.iteration_step, true);
|
||||
twitter.run_iterations(options.iteration_step, true);
|
||||
random.run_iterations(options.iteration_step, true);
|
||||
}
|
||||
} else {
|
||||
for (int iteration = 0; iteration < options.iterations; iteration += options.iteration_step) {
|
||||
if (!options.verbose) { progress.print(iteration); }
|
||||
features.run_iterations(options.iteration_step, false);
|
||||
gsoc_2018.run_iterations(options.iteration_step, false);
|
||||
twitter.run_iterations(options.iteration_step, false);
|
||||
random.run_iterations(options.iteration_step, false);
|
||||
}
|
||||
}
|
||||
if (!options.verbose) { progress.erase(); }
|
||||
|
||||
features.print(options);
|
||||
|
||||
// Gauge effectiveness
|
||||
if (options.verbose) {
|
||||
printf("\n");
|
||||
printf(" Effectiveness Check: Estimated vs. Actual ns/block for real files:\n");
|
||||
printf("\n");
|
||||
printf(" | %8s ", "Stage");
|
||||
printf("| %-15s ", "File");
|
||||
printf("| %11s ", "Est. (Base)");
|
||||
printf("| %11s ", "Est. (Miss)");
|
||||
printf("| %8s ", "Est.");
|
||||
printf("| %8s ", "Actual");
|
||||
printf("| %8s ", "Diff");
|
||||
printf("| %13s ", "Est. Misses");
|
||||
if (features.has_events()) {
|
||||
printf("| %13s ", "Actual Misses");
|
||||
printf("| %13s ", "Diff (Misses)");
|
||||
printf("| %13s ", "Adjusted Miss");
|
||||
printf("| %13s ", "Adjusted Diff");
|
||||
}
|
||||
printf("|\n");
|
||||
printf(" |%.10s", "---------------------------------------");
|
||||
printf("|%.17s", "---------------------------------------");
|
||||
printf("|%.13s", "---------------------------------------");
|
||||
printf("|%.13s", "---------------------------------------");
|
||||
printf("|%.10s", "---------------------------------------");
|
||||
printf("|%.10s", "---------------------------------------");
|
||||
printf("|%.10s", "---------------------------------------");
|
||||
printf("|%.15s", "---------------------------------------");
|
||||
if (features.has_events()) {
|
||||
printf("|%.15s", "---------------------------------------");
|
||||
printf("|%.15s", "---------------------------------------");
|
||||
printf("|%.15s", "---------------------------------------");
|
||||
printf("|%.15s", "---------------------------------------");
|
||||
}
|
||||
printf("|\n");
|
||||
|
||||
options.each_stage([&](auto stage) {
|
||||
print_file_effectiveness(stage, "gsoc-2018.json", gsoc_2018, features);
|
||||
print_file_effectiveness(stage, "twitter.json", twitter, features);
|
||||
print_file_effectiveness(stage, "random.json", random, features);
|
||||
});
|
||||
}
|
||||
|
||||
return EXIT_SUCCESS;
|
||||
}
|
||||
+75
-186
@@ -1,82 +1,7 @@
|
||||
#ifndef _BENCHMARK_H_
|
||||
#define _BENCHMARK_H_
|
||||
#include <float.h>
|
||||
#include <stdint.h>
|
||||
#include <time.h>
|
||||
#ifdef __x86_64__
|
||||
|
||||
const char *unitname = "cycles";
|
||||
|
||||
#define RDTSC_START(cycles) \
|
||||
do { \
|
||||
uint32_t cyc_high, cyc_low; \
|
||||
__asm volatile("cpuid\n" \
|
||||
"rdtsc\n" \
|
||||
"mov %%edx, %0\n" \
|
||||
"mov %%eax, %1" \
|
||||
: "=r"(cyc_high), "=r"(cyc_low) \
|
||||
: \
|
||||
: /* no read only */ \
|
||||
"%rax", "%rbx", "%rcx", "%rdx" /* clobbers */ \
|
||||
); \
|
||||
(cycles) = ((uint64_t)cyc_high << 32) | cyc_low; \
|
||||
} while (0)
|
||||
|
||||
#define RDTSC_STOP(cycles) \
|
||||
do { \
|
||||
uint32_t cyc_high, cyc_low; \
|
||||
__asm volatile("rdtscp\n" \
|
||||
"mov %%edx, %0\n" \
|
||||
"mov %%eax, %1\n" \
|
||||
"cpuid" \
|
||||
: "=r"(cyc_high), "=r"(cyc_low) \
|
||||
: /* no read only registers */ \
|
||||
: "%rax", "%rbx", "%rcx", "%rdx" /* clobbers */ \
|
||||
); \
|
||||
(cycles) = ((uint64_t)cyc_high << 32) | cyc_low; \
|
||||
} while (0)
|
||||
|
||||
#else
|
||||
const char *unitname = " (clock units) ";
|
||||
|
||||
#define RDTSC_START(cycles) \
|
||||
do { \
|
||||
cycles = clock(); \
|
||||
} while (0)
|
||||
|
||||
#define RDTSC_STOP(cycles) \
|
||||
do { \
|
||||
cycles = clock(); \
|
||||
} while (0)
|
||||
#endif
|
||||
|
||||
static __attribute__((noinline)) uint64_t rdtsc_overhead_func(uint64_t dummy) {
|
||||
return dummy;
|
||||
}
|
||||
|
||||
uint64_t global_rdtsc_overhead = (uint64_t)UINT64_MAX;
|
||||
|
||||
#define RDTSC_SET_OVERHEAD(test, repeat) \
|
||||
do { \
|
||||
uint64_t cycles_start, cycles_final, cycles_diff; \
|
||||
uint64_t min_diff = UINT64_MAX; \
|
||||
for (int i = 0; i < repeat; i++) { \
|
||||
__asm volatile("" ::: /* pretend to clobber */ "memory"); \
|
||||
RDTSC_START(cycles_start); \
|
||||
test; \
|
||||
RDTSC_STOP(cycles_final); \
|
||||
cycles_diff = (cycles_final - cycles_start); \
|
||||
if (cycles_diff < min_diff) \
|
||||
min_diff = cycles_diff; \
|
||||
} \
|
||||
global_rdtsc_overhead = min_diff; \
|
||||
} while (0)
|
||||
|
||||
double diff(timespec start, timespec end) {
|
||||
return ((end.tv_nsec + 1000000000 * end.tv_sec) -
|
||||
(start.tv_nsec + 1000000000 * start.tv_sec)) /
|
||||
1000000000.0;
|
||||
}
|
||||
#include "event_counter.h"
|
||||
|
||||
/*
|
||||
* Prints the best number of operations per cycle where
|
||||
@@ -86,135 +11,99 @@ double diff(timespec start, timespec end) {
|
||||
*/
|
||||
#define BEST_TIME(name, test, expected, pre, repeat, size, verbose) \
|
||||
do { \
|
||||
if (global_rdtsc_overhead == UINT64_MAX) { \
|
||||
RDTSC_SET_OVERHEAD(rdtsc_overhead_func(1), repeat); \
|
||||
} \
|
||||
if (verbose) \
|
||||
printf("%-40s\t: ", name); \
|
||||
std::printf("%-40s\t: ", name); \
|
||||
else \
|
||||
printf("\"%s\"\t", name); \
|
||||
std::printf("\"%-40s\"", name); \
|
||||
fflush(NULL); \
|
||||
uint64_t cycles_start, cycles_final, cycles_diff; \
|
||||
uint64_t min_diff = (uint64_t)-1; \
|
||||
double min_sumclockdiff = DBL_MAX; \
|
||||
uint64_t sum_diff = 0; \
|
||||
double sumclockdiff = 0; \
|
||||
struct timespec time1, time2; \
|
||||
for (int i = 0; i < repeat; i++) { \
|
||||
event_collector collector; \
|
||||
event_aggregate aggregate{}; \
|
||||
for (decltype(repeat) i = 0; i < repeat; i++) { \
|
||||
pre; \
|
||||
__asm volatile("" ::: /* pretend to clobber */ "memory"); \
|
||||
clock_gettime(CLOCK_REALTIME, &time1); \
|
||||
RDTSC_START(cycles_start); \
|
||||
std::atomic_thread_fence(std::memory_order_acquire); \
|
||||
collector.start(); \
|
||||
if (test != expected) { \
|
||||
fprintf(stderr, "not expected (%d , %d )", (int)test, (int)expected); \
|
||||
std::fprintf(stderr, "not expected (%d , %d )", (int)test, \
|
||||
(int)expected); \
|
||||
break; \
|
||||
} \
|
||||
RDTSC_STOP(cycles_final); \
|
||||
clock_gettime(CLOCK_REALTIME, &time2); \
|
||||
double thistiming = diff(time1, time2); \
|
||||
sumclockdiff += thistiming; \
|
||||
if (thistiming < min_sumclockdiff) \
|
||||
min_sumclockdiff = thistiming; \
|
||||
cycles_diff = (cycles_final - cycles_start - global_rdtsc_overhead); \
|
||||
if (cycles_diff < min_diff) \
|
||||
min_diff = cycles_diff; \
|
||||
sum_diff += cycles_diff; \
|
||||
std::atomic_thread_fence(std::memory_order_release); \
|
||||
event_count allocate_count = collector.end(); \
|
||||
aggregate << allocate_count; \
|
||||
} \
|
||||
uint64_t S = size; \
|
||||
float cycle_per_op = (min_diff) / (double)S; \
|
||||
float avg_cycle_per_op = (sum_diff) / ((double)S * repeat); \
|
||||
double avg_gb_per_s = \
|
||||
((double)S * repeat) / ((sumclockdiff)*1000.0 * 1000.0 * 1000.0); \
|
||||
double max_gb_per_s = \
|
||||
((double)S) / ((min_sumclockdiff)*1000.0 * 1000.0 * 1000.0); \
|
||||
if (verbose) \
|
||||
printf(" %.3f %s per input byte (best) ", cycle_per_op, unitname); \
|
||||
if (verbose) \
|
||||
printf(" %.3f %s per input byte (avg) ", avg_cycle_per_op, unitname); \
|
||||
if (!verbose) \
|
||||
printf(" %.3f %.3f %.3f %.3f ", cycle_per_op, \
|
||||
avg_cycle_per_op - cycle_per_op, max_gb_per_s, \
|
||||
-avg_gb_per_s + max_gb_per_s); \
|
||||
printf("\n"); \
|
||||
fflush(NULL); \
|
||||
if (collector.has_events()) { \
|
||||
std::printf("%7.3f", \
|
||||
aggregate.best.cycles() / static_cast<double>(size)); \
|
||||
if (verbose) { \
|
||||
std::printf(" cycles/byte "); \
|
||||
} \
|
||||
std::printf("\t"); \
|
||||
std::printf("%7.3f", \
|
||||
aggregate.best.instructions() / static_cast<double>(size)); \
|
||||
if (verbose) { \
|
||||
std::printf(" instructions/byte "); \
|
||||
} \
|
||||
std::printf("\t"); \
|
||||
} \
|
||||
double gb = static_cast<double>(size) / 1000000000.0; \
|
||||
std::printf("%7.3f", gb / aggregate.best.elapsed_sec()); \
|
||||
if (verbose) { \
|
||||
std::printf(" GB/s "); \
|
||||
} \
|
||||
std::printf("\t"); \
|
||||
std::printf("%7.3f", 1.0 / aggregate.best.elapsed_sec()); \
|
||||
if (verbose) { \
|
||||
std::printf(" documents/s "); \
|
||||
} \
|
||||
std::printf("\n"); \
|
||||
std::fflush(NULL); \
|
||||
} while (0)
|
||||
|
||||
// like BEST_TIME, but no check
|
||||
#define BEST_TIME_NOCHECK(name, test, pre, repeat, size, verbose) \
|
||||
do { \
|
||||
if (global_rdtsc_overhead == UINT64_MAX) { \
|
||||
RDTSC_SET_OVERHEAD(rdtsc_overhead_func(1), repeat); \
|
||||
} \
|
||||
if (verbose) \
|
||||
printf("%-40s\t: ", name); \
|
||||
fflush(NULL); \
|
||||
uint64_t cycles_start, cycles_final, cycles_diff; \
|
||||
uint64_t min_diff = (uint64_t)-1; \
|
||||
uint64_t sum_diff = 0; \
|
||||
for (int i = 0; i < repeat; i++) { \
|
||||
std::printf("%-40s\t: ", name); \
|
||||
else \
|
||||
std::printf("\"%-40s\"", name); \
|
||||
std::fflush(NULL); \
|
||||
event_collector collector; \
|
||||
event_aggregate aggregate{}; \
|
||||
for (decltype(repeat) i = 0; i < repeat; i++) { \
|
||||
pre; \
|
||||
__asm volatile("" ::: /* pretend to clobber */ "memory"); \
|
||||
RDTSC_START(cycles_start); \
|
||||
std::atomic_thread_fence(std::memory_order_acquire); \
|
||||
collector.start(); \
|
||||
test; \
|
||||
RDTSC_STOP(cycles_final); \
|
||||
cycles_diff = (cycles_final - cycles_start - global_rdtsc_overhead); \
|
||||
if (cycles_diff < min_diff) \
|
||||
min_diff = cycles_diff; \
|
||||
sum_diff += cycles_diff; \
|
||||
std::atomic_thread_fence(std::memory_order_release); \
|
||||
event_count allocate_count = collector.end(); \
|
||||
aggregate << allocate_count; \
|
||||
} \
|
||||
uint64_t S = size; \
|
||||
float cycle_per_op = (min_diff) / (double)S; \
|
||||
float avg_cycle_per_op = (sum_diff) / ((double)S * repeat); \
|
||||
if (verbose) \
|
||||
printf(" %.3f %s per input byte (best) ", cycle_per_op, unitname); \
|
||||
if (verbose) \
|
||||
printf(" %.3f %s per input byte (avg) ", avg_cycle_per_op, unitname); \
|
||||
if (verbose) \
|
||||
printf("\n"); \
|
||||
if (!verbose) \
|
||||
printf(" %.3f ", cycle_per_op); \
|
||||
fflush(NULL); \
|
||||
} while (0)
|
||||
|
||||
// like BEST_TIME except that we run a function to check the result
|
||||
#define BEST_TIME_CHECK(test, check, pre, repeat, size, verbose) \
|
||||
do { \
|
||||
if (global_rdtsc_overhead == UINT64_MAX) { \
|
||||
RDTSC_SET_OVERHEAD(rdtsc_overhead_func(1), repeat); \
|
||||
} \
|
||||
if (verbose) \
|
||||
printf("%-60s\t:\n", #test); \
|
||||
fflush(NULL); \
|
||||
uint64_t cycles_start, cycles_final, cycles_diff; \
|
||||
uint64_t min_diff = (uint64_t)-1; \
|
||||
uint64_t sum_diff = 0; \
|
||||
for (int i = 0; i < repeat; i++) { \
|
||||
pre; \
|
||||
__asm volatile("" ::: /* pretend to clobber */ "memory"); \
|
||||
RDTSC_START(cycles_start); \
|
||||
test; \
|
||||
RDTSC_STOP(cycles_final); \
|
||||
if (!check) { \
|
||||
printf("error"); \
|
||||
break; \
|
||||
if (collector.has_events()) { \
|
||||
std::printf("%7.3f", \
|
||||
aggregate.best.cycles() / static_cast<double>(size)); \
|
||||
if (verbose) { \
|
||||
std::printf(" cycles/byte "); \
|
||||
} \
|
||||
cycles_diff = (cycles_final - cycles_start - global_rdtsc_overhead); \
|
||||
if (cycles_diff < min_diff) \
|
||||
min_diff = cycles_diff; \
|
||||
sum_diff += cycles_diff; \
|
||||
std::printf("\t"); \
|
||||
std::printf("%7.3f", \
|
||||
aggregate.best.instructions() / static_cast<double>(size)); \
|
||||
if (verbose) { \
|
||||
std::printf(" instructions/byte "); \
|
||||
} \
|
||||
std::printf("\t"); \
|
||||
} \
|
||||
uint64_t S = size; \
|
||||
float cycle_per_op = (min_diff) / (double)S; \
|
||||
float avg_cycle_per_op = (sum_diff) / ((double)S * repeat); \
|
||||
if (verbose) \
|
||||
printf(" %.3f cycles per operation (best) ", cycle_per_op); \
|
||||
if (verbose) \
|
||||
printf("\t%.3f cycles per operation (avg) ", avg_cycle_per_op); \
|
||||
if (verbose) \
|
||||
printf("\n"); \
|
||||
if (!verbose) \
|
||||
printf(" %.3f ", cycle_per_op); \
|
||||
fflush(NULL); \
|
||||
double gb = static_cast<double>(size) / 1000000000.0; \
|
||||
std::printf("%7.3f", gb / aggregate.best.elapsed_sec()); \
|
||||
if (verbose) { \
|
||||
std::printf(" GB/s "); \
|
||||
} \
|
||||
std::printf("\t"); \
|
||||
std::printf("%7.3f", 1.0 / aggregate.best.elapsed_sec()); \
|
||||
if (verbose) { \
|
||||
std::printf(" documents/s "); \
|
||||
} \
|
||||
std::printf("\n"); \
|
||||
std::fflush(NULL); \
|
||||
} while (0)
|
||||
|
||||
#endif
|
||||
|
||||
@@ -0,0 +1,533 @@
|
||||
#ifndef __BENCHMARKER_H
|
||||
#define __BENCHMARKER_H
|
||||
|
||||
#include "event_counter.h"
|
||||
#include "simdjson.h" // For SIMDJSON_DISABLE_DEPRECATED_WARNINGS
|
||||
|
||||
#include <cassert>
|
||||
#include <cctype>
|
||||
#ifndef _MSC_VER
|
||||
#include <dirent.h>
|
||||
#endif
|
||||
#include <unistd.h>
|
||||
#include <cinttypes>
|
||||
|
||||
#include <cstdio>
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
|
||||
#include <algorithm>
|
||||
#include <chrono>
|
||||
#include <cstring>
|
||||
#include <fstream>
|
||||
#include <iomanip>
|
||||
#include <iostream>
|
||||
#include <map>
|
||||
#include <set>
|
||||
#include <sstream>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "linux-perf-events.h"
|
||||
#ifdef __linux__
|
||||
#include <libgen.h>
|
||||
#endif
|
||||
#include "simdjson.h"
|
||||
|
||||
#include <functional>
|
||||
|
||||
using namespace simdjson;
|
||||
using std::cerr;
|
||||
using std::cout;
|
||||
using std::endl;
|
||||
using std::string;
|
||||
using std::to_string;
|
||||
using std::vector;
|
||||
using std::ostream;
|
||||
using std::ofstream;
|
||||
using std::exception;
|
||||
using std::min;
|
||||
using std::max;
|
||||
|
||||
// Initialize "verbose" to go nowhere. We'll read options in main() and set to cout if verbose is true.
|
||||
std::ofstream dev_null;
|
||||
ostream *verbose_stream = &dev_null;
|
||||
const size_t BYTES_PER_BLOCK = 64;
|
||||
|
||||
ostream& verbose() {
|
||||
return *verbose_stream;
|
||||
}
|
||||
|
||||
void exit_error(string message) {
|
||||
cerr << message << endl;
|
||||
exit(EXIT_FAILURE);
|
||||
}
|
||||
|
||||
struct json_stats {
|
||||
size_t bytes = 0;
|
||||
size_t blocks = 0;
|
||||
size_t structurals = 0;
|
||||
size_t blocks_with_utf8 = 0;
|
||||
size_t blocks_with_utf8_flipped = 0;
|
||||
size_t blocks_with_escapes = 0;
|
||||
size_t blocks_with_escapes_flipped = 0;
|
||||
size_t blocks_with_0_structurals = 0;
|
||||
size_t blocks_with_0_structurals_flipped = 0;
|
||||
size_t blocks_with_1_structural = 0;
|
||||
size_t blocks_with_1_structural_flipped = 0;
|
||||
size_t blocks_with_8_structurals = 0;
|
||||
size_t blocks_with_8_structurals_flipped = 0;
|
||||
size_t blocks_with_16_structurals = 0;
|
||||
size_t blocks_with_16_structurals_flipped = 0;
|
||||
|
||||
json_stats(const padded_string& json, const dom::parser& parser) {
|
||||
bytes = json.size();
|
||||
blocks = bytes / BYTES_PER_BLOCK;
|
||||
if (bytes % BYTES_PER_BLOCK > 0) { blocks++; } // Account for remainder block
|
||||
structurals = parser.implementation->n_structural_indexes-1;
|
||||
|
||||
// Calculate stats on blocks that will trigger utf-8 if statements / mispredictions
|
||||
bool last_block_has_utf8 = false;
|
||||
for (size_t block=0; block<blocks; block++) {
|
||||
// Find utf-8 in the block
|
||||
size_t block_start = block*BYTES_PER_BLOCK;
|
||||
size_t block_end = block_start+BYTES_PER_BLOCK;
|
||||
if (block_end > json.size()) { block_end = json.size(); }
|
||||
bool block_has_utf8 = false;
|
||||
for (size_t i=block_start; i<block_end; i++) {
|
||||
if (json.data()[i] & 0x80) {
|
||||
block_has_utf8 = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (block_has_utf8) {
|
||||
blocks_with_utf8++;
|
||||
}
|
||||
if (block > 0 && last_block_has_utf8 != block_has_utf8) {
|
||||
blocks_with_utf8_flipped++;
|
||||
}
|
||||
last_block_has_utf8 = block_has_utf8;
|
||||
}
|
||||
|
||||
// Calculate stats on blocks that will trigger escape if statements / mispredictions
|
||||
bool last_block_has_escapes = false;
|
||||
for (size_t block=0; block<blocks; block++) {
|
||||
// Find utf-8 in the block
|
||||
size_t block_start = block*BYTES_PER_BLOCK;
|
||||
size_t block_end = block_start+BYTES_PER_BLOCK;
|
||||
if (block_end > json.size()) { block_end = json.size(); }
|
||||
bool block_has_escapes = false;
|
||||
for (size_t i=block_start; i<block_end; i++) {
|
||||
if (json.data()[i] == '\\') {
|
||||
block_has_escapes = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (block_has_escapes) {
|
||||
blocks_with_escapes++;
|
||||
}
|
||||
if (block > 0 && last_block_has_escapes != block_has_escapes) {
|
||||
blocks_with_escapes_flipped++;
|
||||
}
|
||||
last_block_has_escapes = block_has_escapes;
|
||||
}
|
||||
|
||||
// Calculate stats on blocks that will trigger structural count if statements / mispredictions
|
||||
bool last_block_has_0_structurals = false;
|
||||
bool last_block_has_1_structural = false;
|
||||
bool last_block_has_8_structurals = false;
|
||||
bool last_block_has_16_structurals = false;
|
||||
size_t structural=0;
|
||||
for (size_t block=0; block<blocks; block++) {
|
||||
// Count structurals in the block
|
||||
int block_structurals=0;
|
||||
while (structural < parser.implementation->n_structural_indexes && parser.implementation->structural_indexes[structural] < (block+1)*BYTES_PER_BLOCK) {
|
||||
block_structurals++;
|
||||
structural++;
|
||||
}
|
||||
|
||||
bool block_has_0_structurals = block_structurals == 0;
|
||||
if (block_has_0_structurals) {
|
||||
blocks_with_0_structurals++;
|
||||
}
|
||||
if (block > 0 && last_block_has_0_structurals != block_has_0_structurals) {
|
||||
blocks_with_0_structurals_flipped++;
|
||||
}
|
||||
last_block_has_0_structurals = block_has_0_structurals;
|
||||
|
||||
bool block_has_1_structural = block_structurals >= 1;
|
||||
if (block_has_1_structural) {
|
||||
blocks_with_1_structural++;
|
||||
}
|
||||
if (block > 0 && last_block_has_1_structural != block_has_1_structural) {
|
||||
blocks_with_1_structural_flipped++;
|
||||
}
|
||||
last_block_has_1_structural = block_has_1_structural;
|
||||
|
||||
bool block_has_8_structurals = block_structurals >= 8;
|
||||
if (block_has_8_structurals) {
|
||||
blocks_with_8_structurals++;
|
||||
}
|
||||
if (block > 0 && last_block_has_8_structurals != block_has_8_structurals) {
|
||||
blocks_with_8_structurals_flipped++;
|
||||
}
|
||||
last_block_has_8_structurals = block_has_8_structurals;
|
||||
|
||||
bool block_has_16_structurals = block_structurals >= 16;
|
||||
if (block_has_16_structurals) {
|
||||
blocks_with_16_structurals++;
|
||||
}
|
||||
if (block > 0 && last_block_has_16_structurals != block_has_16_structurals) {
|
||||
blocks_with_16_structurals_flipped++;
|
||||
}
|
||||
last_block_has_16_structurals = block_has_16_structurals;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
struct progress_bar {
|
||||
int max_value;
|
||||
int total_ticks;
|
||||
double ticks_per_value;
|
||||
int next_tick;
|
||||
progress_bar(int _max_value, int _total_ticks) : max_value(_max_value), total_ticks(_total_ticks), ticks_per_value(double(_total_ticks)/_max_value), next_tick(0) {
|
||||
fprintf(stderr, "[");
|
||||
for (int i=0;i<total_ticks;i++) {
|
||||
fprintf(stderr, " ");
|
||||
}
|
||||
fprintf(stderr, "]");
|
||||
for (int i=0;i<total_ticks+1;i++) {
|
||||
fprintf(stderr, "\b");
|
||||
}
|
||||
}
|
||||
|
||||
void print(int value) {
|
||||
double ticks = value*ticks_per_value;
|
||||
if (ticks >= total_ticks) {
|
||||
ticks = total_ticks-1;
|
||||
}
|
||||
int tick;
|
||||
for (tick=next_tick; tick <= ticks && tick <= total_ticks; tick++) {
|
||||
fprintf(stderr, "=");
|
||||
}
|
||||
next_tick = tick;
|
||||
}
|
||||
void erase() const {
|
||||
for (int i=0;i<next_tick+1;i++) {
|
||||
fprintf(stderr, "\b");
|
||||
}
|
||||
for (int tick=0; tick<=total_ticks+2; tick++) {
|
||||
fprintf(stderr, " ");
|
||||
}
|
||||
for (int tick=0; tick<=total_ticks+2; tick++) {
|
||||
fprintf(stderr, "\b");
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* The speed at which we can allocate memory is strictly system specific.
|
||||
* It depends on the OS and the runtime library. It is subject to various
|
||||
* system-specific knobs. It is not something that we can reasonably
|
||||
* benchmark with crude timings.
|
||||
* If someone wants to optimize how simdjson allocate memory, then it will
|
||||
* almost surely require a distinct benchmarking tool. What is meant by
|
||||
* "memory allocation" also requires a definition. Doing "new char[size]" can
|
||||
* do many different things depending on the system.
|
||||
*/
|
||||
|
||||
enum class BenchmarkStage {
|
||||
ALL, // This excludes allocation
|
||||
ALLOCATE,
|
||||
STAGE1,
|
||||
STAGE2
|
||||
};
|
||||
|
||||
const char* benchmark_stage_name(BenchmarkStage stage) {
|
||||
switch (stage) {
|
||||
case BenchmarkStage::ALL: return "All (Without Allocation)";
|
||||
case BenchmarkStage::ALLOCATE: return "Allocate";
|
||||
case BenchmarkStage::STAGE1: return "Stage 1";
|
||||
case BenchmarkStage::STAGE2: return "Stage 2";
|
||||
default: return "Unknown";
|
||||
}
|
||||
}
|
||||
|
||||
struct benchmarker {
|
||||
// JSON text from loading the file. Owns the memory.
|
||||
padded_string json{};
|
||||
// JSON filename
|
||||
const char *filename;
|
||||
// Event collector that can be turned on to measure cycles, missed branches, etc.
|
||||
event_collector& collector;
|
||||
|
||||
// Statistics about the JSON file independent of its speed (amount of utf-8, structurals, etc.).
|
||||
// Loaded on first parse.
|
||||
json_stats* stats;
|
||||
// Speed and event summary for full parse (stage 1 and stage 2, but *excluding* allocation)
|
||||
event_aggregate all_stages_without_allocation{};
|
||||
// Speed and event summary for stage 1
|
||||
event_aggregate stage1{};
|
||||
// Speed and event summary for stage 2
|
||||
event_aggregate stage2{};
|
||||
// Speed and event summary for allocation
|
||||
event_aggregate allocate_stage{};
|
||||
// Speed and event summary for the repeatly-parsing mode
|
||||
event_aggregate loop{};
|
||||
|
||||
benchmarker(const char *_filename, event_collector& _collector)
|
||||
: filename(_filename), collector(_collector), stats(NULL) {
|
||||
verbose() << "[verbose] loading " << filename << endl;
|
||||
auto error = padded_string::load(filename).get(json);
|
||||
if (error) {
|
||||
exit_error(string("Could not load the file ") + filename);
|
||||
}
|
||||
verbose() << "[verbose] loaded " << filename << endl;
|
||||
}
|
||||
|
||||
~benchmarker() {
|
||||
if (stats) {
|
||||
delete stats;
|
||||
}
|
||||
}
|
||||
|
||||
benchmarker(const benchmarker&) = delete;
|
||||
benchmarker& operator=(const benchmarker&) = delete;
|
||||
|
||||
const event_aggregate& operator[](BenchmarkStage stage) const {
|
||||
switch (stage) {
|
||||
case BenchmarkStage::ALL: return this->all_stages_without_allocation;
|
||||
case BenchmarkStage::STAGE1: return this->stage1;
|
||||
case BenchmarkStage::STAGE2: return this->stage2;
|
||||
case BenchmarkStage::ALLOCATE: return this->allocate_stage;
|
||||
default: exit_error("Unknown stage"); return this->all_stages_without_allocation;
|
||||
}
|
||||
}
|
||||
|
||||
int iterations() const {
|
||||
return all_stages_without_allocation.iterations;
|
||||
}
|
||||
|
||||
simdjson_really_inline void run_iteration(bool stage1_only, bool hotbuffers=false) {
|
||||
// Allocate dom::parser
|
||||
collector.start();
|
||||
dom::parser parser;
|
||||
// We always allocate at least 64KB. Smaller allocations may actually be slower under some systems.
|
||||
error_code error = parser.allocate(json.size() < 65536 ? 65536 : json.size());
|
||||
if (error) {
|
||||
exit_error(string("Unable to allocate_stage ") + to_string(json.size()) + " bytes for the JSON result: " + error_message(error));
|
||||
}
|
||||
event_count allocate_count = collector.end();
|
||||
allocate_stage << allocate_count;
|
||||
// Run it once to get hot buffers
|
||||
if(hotbuffers) {
|
||||
auto result = parser.parse((const uint8_t *)json.data(), json.size());
|
||||
if (result.error()) {
|
||||
exit_error(string("Failed to parse ") + filename + string(":") + error_message(result.error()));
|
||||
}
|
||||
}
|
||||
|
||||
verbose() << "[verbose] allocated memory for parsed JSON " << endl;
|
||||
|
||||
// Stage 1 (find structurals)
|
||||
collector.start();
|
||||
error = parser.implementation->stage1((const uint8_t *)json.data(), json.size(), false);
|
||||
event_count stage1_count = collector.end();
|
||||
stage1 << stage1_count;
|
||||
if (error) {
|
||||
exit_error(string("Failed to parse ") + filename + " during stage 1: " + error_message(error));
|
||||
}
|
||||
|
||||
// Stage 2 (unified machine) and the rest
|
||||
|
||||
if (stage1_only) {
|
||||
all_stages_without_allocation << stage1_count;
|
||||
} else {
|
||||
event_count stage2_count;
|
||||
collector.start();
|
||||
error = parser.implementation->stage2(parser.doc);
|
||||
if (error) {
|
||||
exit_error(string("Failed to parse ") + filename + " during stage 2 parsing " + error_message(error));
|
||||
}
|
||||
stage2_count = collector.end();
|
||||
stage2 << stage2_count;
|
||||
all_stages_without_allocation << stage1_count + stage2_count;
|
||||
}
|
||||
// Calculate stats the first time we parse
|
||||
if (stats == NULL) {
|
||||
if (stage1_only) { // we need stage 2 once
|
||||
error = parser.implementation->stage2(parser.doc);
|
||||
if (error) {
|
||||
printf("Warning: failed to parse during stage 2. Unable to acquire statistics.\n");
|
||||
}
|
||||
}
|
||||
stats = new json_stats(json, parser);
|
||||
}
|
||||
}
|
||||
|
||||
void run_loop(size_t iterations) {
|
||||
dom::parser parser;
|
||||
auto firstresult = parser.parse((const uint8_t *)json.data(), json.size());
|
||||
if (firstresult.error()) {
|
||||
exit_error(string("Failed to parse ") + filename + string(":") + error_message(firstresult.error()));
|
||||
}
|
||||
|
||||
collector.start();
|
||||
// some users want something closer to "number of documents per second"
|
||||
for(size_t i = 0; i < iterations; i++) {
|
||||
auto result = parser.parse((const uint8_t *)json.data(), json.size());
|
||||
if (result.error()) {
|
||||
exit_error(string("Failed to parse ") + filename + string(":") + error_message(result.error()));
|
||||
}
|
||||
}
|
||||
event_count all_loop_count = collector.end();
|
||||
loop << all_loop_count;
|
||||
}
|
||||
|
||||
simdjson_really_inline void run_iterations(size_t iterations, bool stage1_only, bool hotbuffers=false) {
|
||||
for (size_t i = 0; i<iterations; i++) {
|
||||
run_iteration(stage1_only, hotbuffers);
|
||||
}
|
||||
run_loop(iterations);
|
||||
}
|
||||
|
||||
// Gigabyte: https://en.wikipedia.org/wiki/Gigabyte
|
||||
template<typename T>
|
||||
void print_aggregate(const char* prefix, const T& stage) const {
|
||||
printf("%s%-13s: %8.4f ns per block (%6.2f%%) - %8.4f ns per byte - %8.4f ns per structural - %8.4f GB/s\n",
|
||||
prefix,
|
||||
"Speed",
|
||||
stage.elapsed_ns() / static_cast<double>(stats->blocks), // per block
|
||||
percent(stage.elapsed_sec(), all_stages_without_allocation.elapsed_sec()), // %
|
||||
stage.elapsed_ns() / static_cast<double>(stats->bytes), // per byte
|
||||
stage.elapsed_ns() / static_cast<double>(stats->structurals), // per structural
|
||||
(static_cast<double>(json.size()) / 1000000000.0) / stage.elapsed_sec() // GB/s
|
||||
);
|
||||
|
||||
if (collector.has_events()) {
|
||||
printf("%s%-13s: %8.4f per block (%6.2f%%) - %8.4f per byte - %8.4f per structural - %8.3f GHz est. frequency\n",
|
||||
prefix,
|
||||
"Cycles",
|
||||
stage.cycles() / static_cast<double>(stats->blocks),
|
||||
percent(stage.cycles(), all_stages_without_allocation.cycles()),
|
||||
stage.cycles() / static_cast<double>(stats->bytes),
|
||||
stage.cycles() / static_cast<double>(stats->structurals),
|
||||
(stage.cycles() / stage.elapsed_sec()) / 1000000000.0
|
||||
);
|
||||
printf("%s%-13s: %8.4f per block (%6.2f%%) - %8.4f per byte - %8.4f per structural - %8.3f per cycle\n",
|
||||
prefix,
|
||||
"Instructions",
|
||||
stage.instructions() / static_cast<double>(stats->blocks),
|
||||
percent(stage.instructions(), all_stages_without_allocation.instructions()),
|
||||
stage.instructions() / static_cast<double>(stats->bytes),
|
||||
stage.instructions() / static_cast<double>(stats->structurals),
|
||||
stage.instructions() / static_cast<double>(stage.cycles())
|
||||
);
|
||||
|
||||
// NOTE: removed cycles/miss because it is a somewhat misleading stat
|
||||
printf("%s%-13s: %7.0f branch misses (%6.2f%%) - %.0f cache misses (%6.2f%%) - %.2f cache references\n",
|
||||
prefix,
|
||||
"Misses",
|
||||
stage.branch_misses(),
|
||||
percent(stage.branch_misses(), all_stages_without_allocation.branch_misses()),
|
||||
stage.cache_misses(),
|
||||
percent(stage.cache_misses(), all_stages_without_allocation.cache_misses()),
|
||||
stage.cache_references()
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
static double percent(size_t a, size_t b) {
|
||||
return 100.0 * static_cast<double>(a) / static_cast<double>(b);
|
||||
}
|
||||
static double percent(double a, double b) {
|
||||
return 100.0 * a / b;
|
||||
}
|
||||
|
||||
void print(bool tabbed_output) const {
|
||||
if (tabbed_output) {
|
||||
char* filename_copy = (char*)malloc(strlen(filename)+1);
|
||||
SIMDJSON_PUSH_DISABLE_WARNINGS
|
||||
SIMDJSON_DISABLE_DEPRECATED_WARNING // Validated CRT_SECURE safe here
|
||||
strcpy(filename_copy, filename);
|
||||
SIMDJSON_POP_DISABLE_WARNINGS
|
||||
|
||||
#if defined(__linux__)
|
||||
char* base = ::basename(filename_copy);
|
||||
#else
|
||||
char* base = filename_copy;
|
||||
#endif
|
||||
if (strlen(base) >= 5 && !strcmp(base+strlen(base)-5, ".json")) {
|
||||
base[strlen(base)-5] = '\0';
|
||||
}
|
||||
|
||||
double gb = static_cast<double>(json.size()) / 1000000000.0;
|
||||
if (collector.has_events()) {
|
||||
printf("\"%s\"\t%f\t%f\t%f\t%f\t%f\t%f\t%f\n",
|
||||
base,
|
||||
allocate_stage.best.cycles() / static_cast<double>(json.size()),
|
||||
stage1.best.cycles() / static_cast<double>(json.size()),
|
||||
stage2.best.cycles() / static_cast<double>(json.size()),
|
||||
all_stages_without_allocation.best.cycles() / static_cast<double>(json.size()),
|
||||
gb / all_stages_without_allocation.best.elapsed_sec(),
|
||||
gb / stage1.best.elapsed_sec(),
|
||||
gb / stage2.best.elapsed_sec());
|
||||
} else {
|
||||
printf("\"%s\"\t\t\t\t\t%f\t%f\t%f\n",
|
||||
base,
|
||||
gb / all_stages_without_allocation.best.elapsed_sec(),
|
||||
gb / stage1.best.elapsed_sec(),
|
||||
gb / stage2.best.elapsed_sec());
|
||||
}
|
||||
free(filename_copy);
|
||||
} else {
|
||||
printf("\n");
|
||||
printf("%s\n", filename);
|
||||
printf("%s\n", string(strlen(filename), '=').c_str());
|
||||
printf("%9zu blocks - %10zu bytes - %5zu structurals (%5.1f %%)\n", stats->bytes / BYTES_PER_BLOCK, stats->bytes, stats->structurals, percent(stats->structurals, stats->bytes));
|
||||
if (stats) {
|
||||
printf("special blocks with: utf8 %9zu (%5.1f %%) - escape %9zu (%5.1f %%) - 0 structurals %9zu (%5.1f %%) - 1+ structurals %9zu (%5.1f %%) - 8+ structurals %9zu (%5.1f %%) - 16+ structurals %9zu (%5.1f %%)\n",
|
||||
stats->blocks_with_utf8, percent(stats->blocks_with_utf8, stats->blocks),
|
||||
stats->blocks_with_escapes, percent(stats->blocks_with_escapes, stats->blocks),
|
||||
stats->blocks_with_0_structurals, percent(stats->blocks_with_0_structurals, stats->blocks),
|
||||
stats->blocks_with_1_structural, percent(stats->blocks_with_1_structural, stats->blocks),
|
||||
stats->blocks_with_8_structurals, percent(stats->blocks_with_8_structurals, stats->blocks),
|
||||
stats->blocks_with_16_structurals, percent(stats->blocks_with_16_structurals, stats->blocks));
|
||||
printf("special block flips: utf8 %9zu (%5.1f %%) - escape %9zu (%5.1f %%) - 0 structurals %9zu (%5.1f %%) - 1+ structurals %9zu (%5.1f %%) - 8+ structurals %9zu (%5.1f %%) - 16+ structurals %9zu (%5.1f %%)\n",
|
||||
stats->blocks_with_utf8_flipped, percent(stats->blocks_with_utf8_flipped, stats->blocks),
|
||||
stats->blocks_with_escapes_flipped, percent(stats->blocks_with_escapes_flipped, stats->blocks),
|
||||
stats->blocks_with_0_structurals_flipped, percent(stats->blocks_with_0_structurals_flipped, stats->blocks),
|
||||
stats->blocks_with_1_structural_flipped, percent(stats->blocks_with_1_structural_flipped, stats->blocks),
|
||||
stats->blocks_with_8_structurals_flipped, percent(stats->blocks_with_8_structurals_flipped, stats->blocks),
|
||||
stats->blocks_with_16_structurals_flipped, percent(stats->blocks_with_16_structurals_flipped, stats->blocks));
|
||||
}
|
||||
printf("\n");
|
||||
printf("All Stages (excluding allocation)\n");
|
||||
print_aggregate("| " , all_stages_without_allocation.best);
|
||||
// frequently, allocation is a tiny fraction of the running time so we omit it
|
||||
if(allocate_stage.best.elapsed_sec() > 0.01 * all_stages_without_allocation.best.elapsed_sec()) {
|
||||
printf("|- Allocation\n");
|
||||
print_aggregate("| ", allocate_stage.best);
|
||||
}
|
||||
printf("|- Stage 1\n");
|
||||
print_aggregate("| ", stage1.best);
|
||||
printf("|- Stage 2\n");
|
||||
print_aggregate("| ", stage2.best);
|
||||
if (collector.has_events()) {
|
||||
double freq1 = (stage1.best.cycles() / stage1.best.elapsed_sec()) / 1000000000.0;
|
||||
double freq2 = (stage2.best.cycles() / stage2.best.elapsed_sec()) / 1000000000.0;
|
||||
double freqall = (all_stages_without_allocation.best.cycles() / all_stages_without_allocation.best.elapsed_sec()) / 1000000000.0;
|
||||
double freqmin = min(freq1, freq2);
|
||||
double freqmax = max(freq1, freq2);
|
||||
if((freqall < 0.95 * freqmin) or (freqall > 1.05 * freqmax)) {
|
||||
printf("\nWarning: The processor frequency fluctuates in an expected way!!!\n"
|
||||
"Range for stage 1 and stage 2 : [%.3f GHz, %.3f GHz], overall: %.3f GHz.\n",
|
||||
freqmin, freqmax, freqall);
|
||||
}
|
||||
}
|
||||
printf("\n%.1f documents parsed per second (best)\n", 1.0/static_cast<double>(all_stages_without_allocation.best.elapsed_sec()));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,101 @@
|
||||
# Relevant targets:
|
||||
# checkperf-parse: builds the reference checkperf-parse, syncing reference repository if needed
|
||||
# checkperf: builds the targets needed for checkperf (parse, perfdiff, checkperf-parse)
|
||||
# update-checkperf-repo: updates the reference repository we're checking performance against
|
||||
# checkperf-repo: initialize and sync reference repository (first time only)
|
||||
# TEST checkperf: runs the actual checkperf test
|
||||
|
||||
# Clone the repository if it's not there
|
||||
find_package(Git QUIET)
|
||||
if (Git_FOUND AND (GIT_VERSION_STRING VERSION_GREATER "2.1.4") AND (NOT CMAKE_GENERATOR MATCHES Ninja) ) # We use "-C" which requires a recent git
|
||||
message(STATUS "Git is available and it is recent. We are enabling checkperf targets.")
|
||||
# sync_git_repository(myrepo ...) creates two targets:
|
||||
# myrepo - if the repo does not exist, creates and syncs it against the origin branch
|
||||
# update_myrepo - will update the repo against the origin branch (and create if needed)
|
||||
function(sync_git_repository name dir remote branch url)
|
||||
# This conditionally creates the git repository
|
||||
add_custom_command(
|
||||
OUTPUT ${dir}/.git/config
|
||||
COMMAND ${GIT_EXECUTABLE} init ${dir}
|
||||
COMMAND ${GIT_EXECUTABLE} -C ${dir} remote add ${remote} ${url}
|
||||
)
|
||||
add_custom_target(init-${name} DEPENDS ${dir}/.git/config)
|
||||
# This conditionally syncs the git repository, first time only
|
||||
add_custom_command(
|
||||
OUTPUT ${dir}/.git/FETCH_HEAD
|
||||
COMMAND ${GIT_EXECUTABLE} remote set-url ${remote} ${url}
|
||||
COMMAND ${GIT_EXECUTABLE} fetch --depth=1 ${remote} ${branch}
|
||||
COMMAND ${GIT_EXECUTABLE} reset --hard ${remote}/${branch}
|
||||
WORKING_DIRECTORY ${dir}
|
||||
DEPENDS init-${name}
|
||||
)
|
||||
# This is the ${name} target, which will create and sync the repo first time only
|
||||
add_custom_target(${name} DEPENDS ${dir}/.git/FETCH_HEAD)
|
||||
# This is the update-${name} target, which will sync the repo (creating it if needed)
|
||||
add_custom_target(
|
||||
update-${name}
|
||||
COMMAND ${GIT_EXECUTABLE} remote set-url ${remote} ${url}
|
||||
COMMAND ${GIT_EXECUTABLE} fetch --depth=1 ${remote} ${branch}
|
||||
COMMAND ${GIT_EXECUTABLE} reset --hard ${remote}/${branch}
|
||||
WORKING_DIRECTORY ${dir}
|
||||
DEPENDS init-${name}
|
||||
)
|
||||
endfunction(sync_git_repository)
|
||||
|
||||
set(SIMDJSON_CHECKPERF_REMOTE origin CACHE STRING "Remote repository to compare performance against")
|
||||
set(SIMDJSON_CHECKPERF_BRANCH master CACHE STRING "Branch to compare performance against")
|
||||
set(SIMDJSON_CHECKPERF_DIR ${CMAKE_CURRENT_BINARY_DIR}/checkperf-reference/${SIMDJSON_CHECKPERF_BRANCH} CACHE STRING "Location to put checkperf performance comparison repository")
|
||||
set(SIMDJSON_CHECKPERF_ARGS ${EXAMPLE_JSON} CACHE STRING "Arguments to pass to parse during checkperf")
|
||||
sync_git_repository(checkperf-repo ${SIMDJSON_CHECKPERF_DIR} ${SIMDJSON_CHECKPERF_REMOTE} ${SIMDJSON_CHECKPERF_BRANCH} ${SIMDJSON_GITHUB_REPOSITORY})
|
||||
|
||||
# Commands to cause cmake on benchmark/checkperf-master/build/
|
||||
# - first, copy CMakeCache.txt
|
||||
add_custom_command(
|
||||
OUTPUT ${SIMDJSON_CHECKPERF_DIR}/build/CMakeCache.txt
|
||||
COMMAND ${CMAKE_COMMAND} -E make_directory ${SIMDJSON_CHECKPERF_DIR}/build
|
||||
COMMAND ${CMAKE_COMMAND} -E copy ${SIMDJSON_USER_CMAKECACHE} ${SIMDJSON_CHECKPERF_DIR}/build/CMakeCache.txt
|
||||
DEPENDS checkperf-repo simdjson-user-cmakecache
|
||||
)
|
||||
# - second, cmake ..
|
||||
add_custom_command(
|
||||
OUTPUT ${SIMDJSON_CHECKPERF_DIR}/build/cmake_install.cmake # We make many things but this seems the most cross-platform one we can depend on
|
||||
COMMAND
|
||||
${CMAKE_COMMAND} -E env CXX=${CMAKE_CXX_COMPILER} CC=${CMAKE_C_COMPILER}
|
||||
${CMAKE_COMMAND} -DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE} -DSIMDJSON_GOOGLE_BENCHMARKS=OFF -DSIMDJSON_COMPETITION=OFF -G ${CMAKE_GENERATOR} ..
|
||||
WORKING_DIRECTORY ${SIMDJSON_CHECKPERF_DIR}/build
|
||||
DEPENDS ${SIMDJSON_CHECKPERF_DIR}/build/CMakeCache.txt
|
||||
)
|
||||
|
||||
# - third, build parse.
|
||||
if (CMAKE_CONFIGURATION_TYPES)
|
||||
set(CHECKPERF_PARSE ${SIMDJSON_CHECKPERF_DIR}/build/benchmark/$<CONFIGURATION>/parse)
|
||||
else()
|
||||
set(CHECKPERF_PARSE ${SIMDJSON_CHECKPERF_DIR}/build/benchmark/parse)
|
||||
endif()
|
||||
add_custom_target(
|
||||
checkperf-parse ALL # TODO is ALL necessary?
|
||||
# Build parse
|
||||
COMMAND ${CMAKE_COMMAND} --build . --target parse --config $<CONFIGURATION>
|
||||
WORKING_DIRECTORY ${SIMDJSON_CHECKPERF_DIR}/build
|
||||
DEPENDS ${SIMDJSON_CHECKPERF_DIR}/build/cmake_install.cmake # We make many things but this seems the most cross-platform one we can depend on
|
||||
)
|
||||
|
||||
# Target to build everything needed for the checkperf test
|
||||
add_custom_target(checkperf DEPENDS parse perfdiff checkperf-parse)
|
||||
|
||||
# Add the actual checkperf test
|
||||
add_test(
|
||||
NAME checkperf
|
||||
# COMMAND ECHO $<TARGET_FILE:perfdiff> \"$<TARGET_FILE:parse> -t ${SIMDJSON_CHECKPERF_ARGS}\" \"${CHECKPERF_PARSE} -t ${SIMDJSON_CHECKPERF_ARGS}\" }
|
||||
COMMAND $<TARGET_FILE:perfdiff> $<TARGET_FILE:parse> ${CHECKPERF_PARSE} -H -t ${SIMDJSON_CHECKPERF_ARGS}
|
||||
)
|
||||
set_property(TEST checkperf APPEND PROPERTY LABELS per_implementation)
|
||||
set_property(TEST checkperf APPEND PROPERTY DEPENDS parse perfdiff ${SIMDJSON_USER_CMAKECACHE})
|
||||
set_property(TEST checkperf PROPERTY RUN_SERIAL TRUE)
|
||||
else()
|
||||
if (CMAKE_GENERATOR MATCHES Ninja)
|
||||
message(STATUS "We disable the checkperf targets under Ninja.")
|
||||
else()
|
||||
message(STATUS "Either git is unavailable or else it is too old. We are disabling checkperf targets.")
|
||||
endif()
|
||||
endif ()
|
||||
@@ -0,0 +1,52 @@
|
||||
|
||||
#pragma once
|
||||
#include <vector>
|
||||
#include <cstdint>
|
||||
#include "event_counter.h"
|
||||
#include "json_benchmark.h"
|
||||
|
||||
|
||||
bool equals(const char *s1, const char *s2) { return strcmp(s1, s2) == 0; }
|
||||
|
||||
void remove_duplicates(std::vector<int64_t> &v) {
|
||||
std::sort(v.begin(), v.end());
|
||||
auto last = std::unique(v.begin(), v.end());
|
||||
v.erase(last, v.end());
|
||||
}
|
||||
|
||||
//
|
||||
// Interface
|
||||
//
|
||||
|
||||
namespace distinct_user_id {
|
||||
template<typename T> static void DistinctUserID(benchmark::State &state);
|
||||
} // namespace
|
||||
|
||||
//
|
||||
// Implementation
|
||||
//
|
||||
|
||||
#include "dom.h"
|
||||
|
||||
|
||||
namespace distinct_user_id {
|
||||
|
||||
using namespace simdjson;
|
||||
|
||||
template<typename T> static void DistinctUserID(benchmark::State &state) {
|
||||
//
|
||||
// Load the JSON file
|
||||
//
|
||||
constexpr const char *TWITTER_JSON = SIMDJSON_BENCHMARK_DATA_DIR "twitter.json";
|
||||
error_code error;
|
||||
padded_string json;
|
||||
if ((error = padded_string::load(TWITTER_JSON).get(json))) {
|
||||
std::cerr << error << std::endl;
|
||||
state.SkipWithError("error loading");
|
||||
return;
|
||||
}
|
||||
|
||||
JsonBenchmark<T, Dom>(state, json);
|
||||
}
|
||||
|
||||
} // namespace distinct_user_id
|
||||
@@ -0,0 +1,89 @@
|
||||
#pragma once
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
|
||||
#include "distinctuserid.h"
|
||||
|
||||
namespace distinct_user_id {
|
||||
|
||||
using namespace simdjson;
|
||||
|
||||
|
||||
simdjson_really_inline void simdjson_recurse(std::vector<int64_t> & v, simdjson::dom::element element);
|
||||
void simdjson_recurse(std::vector<int64_t> & v, simdjson::dom::array array) {
|
||||
for (auto child : array) {
|
||||
simdjson_recurse(v, child);
|
||||
}
|
||||
}
|
||||
void simdjson_recurse(std::vector<int64_t> & v, simdjson::dom::object object) {
|
||||
for (auto [key, value] : object) {
|
||||
if((key.size() == 4) && (memcmp(key.data(), "user", 4) == 0)) {
|
||||
// we are in an object under the key "user"
|
||||
simdjson::error_code error;
|
||||
simdjson::dom::object child_object;
|
||||
simdjson::dom::object child_array;
|
||||
if (not (error = value.get(child_object))) {
|
||||
for (auto [child_key, child_value] : child_object) {
|
||||
if((child_key.size() == 2) && (memcmp(child_key.data(), "id", 2) == 0)) {
|
||||
int64_t x;
|
||||
if (not (error = child_value.get(x))) {
|
||||
v.push_back(x);
|
||||
}
|
||||
}
|
||||
simdjson_recurse(v, child_value);
|
||||
}
|
||||
} else if (not (error = value.get(child_array))) {
|
||||
simdjson_recurse(v, child_array);
|
||||
}
|
||||
// end of: we are in an object under the key "user"
|
||||
} else {
|
||||
simdjson_recurse(v, value);
|
||||
}
|
||||
}
|
||||
}
|
||||
simdjson_really_inline void simdjson_recurse(std::vector<int64_t> & v, simdjson::dom::element element) {
|
||||
simdjson_unused simdjson::error_code error;
|
||||
simdjson::dom::array array;
|
||||
simdjson::dom::object object;
|
||||
if (not (error = element.get(array))) {
|
||||
simdjson_recurse(v, array);
|
||||
} else if (not (error = element.get(object))) {
|
||||
simdjson_recurse(v, object);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
class Dom {
|
||||
public:
|
||||
simdjson_really_inline bool Run(const padded_string &json);
|
||||
simdjson_really_inline const std::vector<int64_t> &Result() { return ids; }
|
||||
simdjson_really_inline size_t ItemCount() { return ids.size(); }
|
||||
|
||||
private:
|
||||
dom::parser parser{};
|
||||
std::vector<int64_t> ids{};
|
||||
|
||||
};
|
||||
void print_vec(const std::vector<int64_t> &v) {
|
||||
for (auto i : v) {
|
||||
std::cout << i << " ";
|
||||
}
|
||||
std::cout << std::endl;
|
||||
}
|
||||
|
||||
simdjson_really_inline bool Dom::Run(const padded_string &json) {
|
||||
ids.clear();
|
||||
dom::element doc = parser.parse(json);
|
||||
simdjson_recurse(ids, doc);
|
||||
remove_duplicates(ids);
|
||||
return true;
|
||||
}
|
||||
|
||||
BENCHMARK_TEMPLATE(DistinctUserID, Dom);
|
||||
|
||||
} // namespace distinct_user_id
|
||||
|
||||
#endif // SIMDJSON_EXCEPTIONS
|
||||
@@ -0,0 +1,67 @@
|
||||
#pragma once
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
|
||||
#include "distinctuserid.h"
|
||||
|
||||
namespace distinct_user_id {
|
||||
|
||||
using namespace simdjson;
|
||||
using namespace simdjson::builtin;
|
||||
|
||||
|
||||
class OnDemand {
|
||||
public:
|
||||
OnDemand() {
|
||||
if(!displayed_implementation) {
|
||||
std::cout << "On Demand implementation: " << builtin_implementation()->name() << std::endl;
|
||||
displayed_implementation = true;
|
||||
}
|
||||
}
|
||||
simdjson_really_inline bool Run(const padded_string &json);
|
||||
simdjson_really_inline const std::vector<int64_t> &Result() { return ids; }
|
||||
simdjson_really_inline size_t ItemCount() { return ids.size(); }
|
||||
|
||||
private:
|
||||
ondemand::parser parser{};
|
||||
std::vector<int64_t> ids{};
|
||||
|
||||
static inline bool displayed_implementation = false;
|
||||
};
|
||||
|
||||
simdjson_really_inline bool OnDemand::Run(const padded_string &json) {
|
||||
ids.clear();
|
||||
// Walk the document, parsing as we go
|
||||
auto doc = parser.iterate(json);
|
||||
for (ondemand::object tweet : doc["statuses"]) {
|
||||
// We believe that all statuses have a matching
|
||||
// user, and we are willing to throw when they do not:
|
||||
//
|
||||
// You might think that you do not need the braces, but
|
||||
// you do, otherwise you will get the wrong answer. That is
|
||||
// because you can only have one active object or array
|
||||
// at a time.
|
||||
{
|
||||
ondemand::object user = tweet["user"];
|
||||
int64_t id = user["id"];
|
||||
ids.push_back(id);
|
||||
}
|
||||
// Not all tweets have a "retweeted_status", but when they do
|
||||
// we want to go and find the user within.
|
||||
auto retweet = tweet["retweeted_status"];
|
||||
if(!retweet.error()) {
|
||||
ondemand::object retweet_content = retweet;
|
||||
ondemand::object reuser = retweet_content["user"];
|
||||
int64_t rid = reuser["id"];
|
||||
ids.push_back(rid);
|
||||
}
|
||||
}
|
||||
remove_duplicates(ids);
|
||||
return true;
|
||||
}
|
||||
|
||||
BENCHMARK_TEMPLATE(DistinctUserID, OnDemand);
|
||||
|
||||
} // namespace distinct_user_id
|
||||
|
||||
#endif // SIMDJSON_EXCEPTIONS
|
||||
@@ -1,9 +1,13 @@
|
||||
#include "simdjson/jsonparser.h"
|
||||
#include "simdjson.h"
|
||||
#include <algorithm>
|
||||
#include <cstring>
|
||||
#include <unistd.h>
|
||||
#include <vector>
|
||||
|
||||
#include "benchmark.h"
|
||||
|
||||
SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
|
||||
|
||||
// #define RAPIDJSON_SSE2 // bad for performance
|
||||
// #define RAPIDJSON_SSE42 // bad for performance
|
||||
#include "rapidjson/document.h"
|
||||
@@ -13,79 +17,124 @@
|
||||
|
||||
#include "sajson.h"
|
||||
|
||||
SIMDJSON_POP_DISABLE_WARNINGS
|
||||
|
||||
using namespace rapidjson;
|
||||
using namespace std;
|
||||
|
||||
bool equals(const char *s1, const char *s2) { return strcmp(s1, s2) == 0; }
|
||||
|
||||
void remove_duplicates(vector<int64_t> &v) {
|
||||
void remove_duplicates(std::vector<int64_t> &v) {
|
||||
std::sort(v.begin(), v.end());
|
||||
auto last = std::unique(v.begin(), v.end());
|
||||
v.erase(last, v.end());
|
||||
}
|
||||
|
||||
void print_vec(vector<int64_t> &v) {
|
||||
void print_vec(const std::vector<int64_t> &v) {
|
||||
for (auto i : v) {
|
||||
std::cout << i << " ";
|
||||
}
|
||||
std::cout << std::endl;
|
||||
}
|
||||
|
||||
void simdjson_traverse(std::vector<int64_t> &answer, ParsedJson::iterator &i) {
|
||||
switch (i.get_type()) {
|
||||
case '{':
|
||||
if (i.down()) {
|
||||
do {
|
||||
bool founduser = equals(i.get_string(), "user");
|
||||
i.next(); // move to value
|
||||
if (i.is_object()) {
|
||||
if (founduser && i.move_to_key("id")) {
|
||||
if (i.is_integer()) {
|
||||
answer.push_back(i.get_integer());
|
||||
// clang-format off
|
||||
|
||||
// simdjson_recurse below can be implemented like so but it is slow:
|
||||
/*void simdjson_recurse(std::vector<int64_t> & v, simdjson::dom::element element) {
|
||||
error_code error;
|
||||
if (element.is_array()) {
|
||||
dom::array array;
|
||||
error = element.get(array);
|
||||
for (auto child : array) {
|
||||
if (child.is<simdjson::dom::array>() || child.is<simdjson::dom::object>()) {
|
||||
simdjson_recurse(v, child);
|
||||
}
|
||||
}
|
||||
} else if (element.is_object()) {
|
||||
int64_t id;
|
||||
error = element["user"]["id"].get(id);
|
||||
if(!error) {
|
||||
v.push_back(id);
|
||||
}
|
||||
for (auto [key, value] : object) {
|
||||
if (value.is<simdjson::dom::array>() || value.is<simdjson::dom::object>()) {
|
||||
simdjson_recurse(v, value);
|
||||
}
|
||||
}
|
||||
}
|
||||
}*/
|
||||
// clang-format on
|
||||
|
||||
|
||||
simdjson_really_inline void simdjson_recurse(std::vector<int64_t> & v, simdjson::dom::element element);
|
||||
void simdjson_recurse(std::vector<int64_t> & v, simdjson::dom::array array) {
|
||||
for (auto child : array) {
|
||||
simdjson_recurse(v, child);
|
||||
}
|
||||
}
|
||||
void simdjson_recurse(std::vector<int64_t> & v, simdjson::dom::object object) {
|
||||
for (auto [key, value] : object) {
|
||||
if((key.size() == 4) && (memcmp(key.data(), "user", 4) == 0)) {
|
||||
// we are in an object under the key "user"
|
||||
simdjson::error_code error;
|
||||
simdjson::dom::object child_object;
|
||||
simdjson::dom::object child_array;
|
||||
if (not (error = value.get(child_object))) {
|
||||
for (auto [child_key, child_value] : child_object) {
|
||||
if((child_key.size() == 2) && (memcmp(child_key.data(), "id", 2) == 0)) {
|
||||
int64_t x;
|
||||
if (not (error = child_value.get(x))) {
|
||||
v.push_back(x);
|
||||
}
|
||||
i.up();
|
||||
}
|
||||
simdjson_traverse(answer, i);
|
||||
} else if (i.is_array()) {
|
||||
simdjson_traverse(answer, i);
|
||||
simdjson_recurse(v, child_value);
|
||||
}
|
||||
} while (i.next());
|
||||
i.up();
|
||||
} else if (not (error = value.get(child_array))) {
|
||||
simdjson_recurse(v, child_array);
|
||||
}
|
||||
// end of: we are in an object under the key "user"
|
||||
} else {
|
||||
simdjson_recurse(v, value);
|
||||
}
|
||||
break;
|
||||
case '[':
|
||||
if (i.down()) {
|
||||
do {
|
||||
if (i.is_object_or_array()) {
|
||||
simdjson_traverse(answer, i);
|
||||
}
|
||||
} while (i.next());
|
||||
i.up();
|
||||
}
|
||||
break;
|
||||
case 'l':
|
||||
case 'd':
|
||||
case 'n':
|
||||
case 't':
|
||||
case 'f':
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
simdjson_really_inline void simdjson_recurse(std::vector<int64_t> & v, simdjson::dom::element element) {
|
||||
simdjson_unused simdjson::error_code error;
|
||||
simdjson::dom::array array;
|
||||
simdjson::dom::object object;
|
||||
if (not (error = element.get(array))) {
|
||||
simdjson_recurse(v, array);
|
||||
} else if (not (error = element.get(object))) {
|
||||
simdjson_recurse(v, object);
|
||||
}
|
||||
}
|
||||
|
||||
std::vector<int64_t> simdjson_computestats(const std::string_view &p) {
|
||||
simdjson_really_inline std::vector<int64_t>
|
||||
simdjson_just_dom(simdjson::dom::element doc) {
|
||||
std::vector<int64_t> answer;
|
||||
ParsedJson pj = build_parsed_json(p);
|
||||
if (!pj.isValid()) {
|
||||
return answer;
|
||||
}
|
||||
ParsedJson::iterator i(pj);
|
||||
|
||||
simdjson_traverse(answer, i);
|
||||
simdjson_recurse(answer, doc);
|
||||
remove_duplicates(answer);
|
||||
return answer;
|
||||
}
|
||||
|
||||
simdjson_really_inline std::vector<int64_t>
|
||||
simdjson_compute_stats(const simdjson::padded_string &p) {
|
||||
std::vector<int64_t> answer;
|
||||
simdjson::dom::parser parser;
|
||||
simdjson::dom::element doc;
|
||||
auto error = parser.parse(p).get(doc);
|
||||
if (!error) {
|
||||
simdjson_recurse(answer, doc);
|
||||
remove_duplicates(answer);
|
||||
}
|
||||
return answer;
|
||||
}
|
||||
|
||||
simdjson_really_inline simdjson::error_code
|
||||
simdjson_just_parse(const simdjson::padded_string &p) {
|
||||
simdjson::dom::parser parser;
|
||||
return parser.parse(p).error();
|
||||
}
|
||||
|
||||
void sajson_traverse(std::vector<int64_t> &answer, const sajson::value &node) {
|
||||
using namespace sajson;
|
||||
switch (node.get_type()) {
|
||||
@@ -98,22 +147,26 @@ void sajson_traverse(std::vector<int64_t> &answer, const sajson::value &node) {
|
||||
}
|
||||
case TYPE_OBJECT: {
|
||||
auto length = node.get_length();
|
||||
// sajson has O(log n) find_object_key, but we still visit each node anyhow
|
||||
// because we need to visit all values.
|
||||
for (auto i = 0u; i < length; ++i) {
|
||||
if (equals(node.get_object_key(i).data(), "user")) { // found a user!!!
|
||||
auto uservalue = node.get_object_value(i); // get the value
|
||||
if (uservalue.get_type() ==
|
||||
auto key = node.get_object_key(i); // expected: sajson::string
|
||||
bool found_user =
|
||||
(key.length() == 4) && (memcmp(key.data(), "user", 4) == 0);
|
||||
if (found_user) { // found a user!!!
|
||||
auto user_value = node.get_object_value(i); // get the value
|
||||
if (user_value.get_type() ==
|
||||
TYPE_OBJECT) { // the value should be an object
|
||||
auto uservaluelength = uservalue.get_length();
|
||||
for (auto j = 0u; j < uservaluelength;
|
||||
++j) { // go through the children
|
||||
if (equals(uservalue.get_object_key(j).data(),
|
||||
"id")) { // ah ah found id
|
||||
auto v = uservalue.get_object_value(j);
|
||||
if (v.get_type() == TYPE_INTEGER) { // check that it is an integer
|
||||
answer.push_back(v.get_integer_value()); // record it!
|
||||
} else if (v.get_type() == TYPE_DOUBLE) {
|
||||
answer.push_back((int64_t)v.get_double_value()); // record it!
|
||||
}
|
||||
// now we know that we only need one value
|
||||
auto user_value_length = user_value.get_length();
|
||||
auto right_index =
|
||||
user_value.find_object_key(sajson::string("id", 2));
|
||||
if (right_index < user_value_length) {
|
||||
auto v = user_value.get_object_value(right_index);
|
||||
if (v.get_type() == TYPE_INTEGER) { // check that it is an integer
|
||||
answer.push_back(v.get_integer_value()); // record it!
|
||||
} else if (v.get_type() == TYPE_DOUBLE) {
|
||||
answer.push_back((int64_t)v.get_double_value()); // record it!
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -134,13 +187,23 @@ void sajson_traverse(std::vector<int64_t> &answer, const sajson::value &node) {
|
||||
}
|
||||
}
|
||||
|
||||
std::vector<int64_t> sasjon_computestats(const std::string_view &p) {
|
||||
simdjson_really_inline std::vector<int64_t>
|
||||
sasjon_just_dom(sajson::document &d) {
|
||||
std::vector<int64_t> answer;
|
||||
sajson_traverse(answer, d.get_root());
|
||||
remove_duplicates(answer);
|
||||
return answer;
|
||||
}
|
||||
|
||||
simdjson_really_inline std::vector<int64_t>
|
||||
sasjon_compute_stats(const simdjson::padded_string &p) {
|
||||
std::vector<int64_t> answer;
|
||||
char *buffer = (char *)malloc(p.size());
|
||||
memcpy(buffer, p.data(), p.size());
|
||||
auto d = sajson::parse(sajson::dynamic_allocation(),
|
||||
sajson::mutable_string_view(p.size(), buffer));
|
||||
if (!d.is_valid()) {
|
||||
free(buffer);
|
||||
return answer;
|
||||
}
|
||||
sajson_traverse(answer, d.get_root());
|
||||
@@ -149,12 +212,25 @@ std::vector<int64_t> sasjon_computestats(const std::string_view &p) {
|
||||
return answer;
|
||||
}
|
||||
|
||||
simdjson_really_inline bool
|
||||
sasjon_just_parse(const simdjson::padded_string &p) {
|
||||
char *buffer = (char *)malloc(p.size());
|
||||
memcpy(buffer, p.data(), p.size());
|
||||
auto d = sajson::parse(sajson::dynamic_allocation(),
|
||||
sajson::mutable_string_view(p.size(), buffer));
|
||||
bool answer = !d.is_valid();
|
||||
free(buffer);
|
||||
return answer;
|
||||
}
|
||||
|
||||
void rapid_traverse(std::vector<int64_t> &answer, const rapidjson::Value &v) {
|
||||
switch (v.GetType()) {
|
||||
case kObjectType:
|
||||
for (Value::ConstMemberIterator m = v.MemberBegin(); m != v.MemberEnd();
|
||||
++m) {
|
||||
if (equals(m->name.GetString(), "user")) {
|
||||
bool found_user = (m->name.GetStringLength() == 4) &&
|
||||
(memcmp(m->name.GetString(), "user", 4) == 0);
|
||||
if (found_user) {
|
||||
const rapidjson::Value &child = m->value;
|
||||
if (child.GetType() == kObjectType) {
|
||||
for (Value::ConstMemberIterator k = child.MemberBegin();
|
||||
@@ -187,7 +263,16 @@ void rapid_traverse(std::vector<int64_t> &answer, const rapidjson::Value &v) {
|
||||
}
|
||||
}
|
||||
|
||||
std::vector<int64_t> rapid_computestats(const std::string_view &p) {
|
||||
simdjson_really_inline std::vector<int64_t>
|
||||
rapid_just_dom(rapidjson::Document &d) {
|
||||
std::vector<int64_t> answer;
|
||||
rapid_traverse(answer, d);
|
||||
remove_duplicates(answer);
|
||||
return answer;
|
||||
}
|
||||
|
||||
simdjson_really_inline std::vector<int64_t>
|
||||
rapid_compute_stats(const simdjson::padded_string &p) {
|
||||
std::vector<int64_t> answer;
|
||||
char *buffer = (char *)malloc(p.size() + 1);
|
||||
memcpy(buffer, p.data(), p.size());
|
||||
@@ -195,6 +280,7 @@ std::vector<int64_t> rapid_computestats(const std::string_view &p) {
|
||||
rapidjson::Document d;
|
||||
d.ParseInsitu<kParseValidateEncodingFlag>(buffer);
|
||||
if (d.HasParseError()) {
|
||||
free(buffer);
|
||||
return answer;
|
||||
}
|
||||
rapid_traverse(answer, d);
|
||||
@@ -203,15 +289,27 @@ std::vector<int64_t> rapid_computestats(const std::string_view &p) {
|
||||
return answer;
|
||||
}
|
||||
|
||||
simdjson_really_inline bool
|
||||
rapid_just_parse(const simdjson::padded_string &p) {
|
||||
char *buffer = (char *)malloc(p.size() + 1);
|
||||
memcpy(buffer, p.data(), p.size());
|
||||
buffer[p.size()] = '\0';
|
||||
rapidjson::Document d;
|
||||
d.ParseInsitu<kParseValidateEncodingFlag>(buffer);
|
||||
bool answer = d.HasParseError();
|
||||
free(buffer);
|
||||
return answer;
|
||||
}
|
||||
|
||||
int main(int argc, char *argv[]) {
|
||||
bool verbose = false;
|
||||
bool justdata = false;
|
||||
bool just_data = false;
|
||||
|
||||
int c;
|
||||
while ((c = getopt(argc, argv, "vt")) != -1)
|
||||
switch (c) {
|
||||
case 't':
|
||||
justdata = true;
|
||||
just_data = true;
|
||||
break;
|
||||
case 'v':
|
||||
verbose = true;
|
||||
@@ -220,45 +318,47 @@ int main(int argc, char *argv[]) {
|
||||
abort();
|
||||
}
|
||||
if (optind >= argc) {
|
||||
cerr << "Using different parsers, we compute the content statistics of "
|
||||
"JSON documents.\n";
|
||||
cerr << "Usage: " << argv[0] << " <jsonfile>\n";
|
||||
cerr << "Or " << argv[0] << " -v <jsonfile>\n";
|
||||
std::cerr
|
||||
<< "Using different parsers, we compute the content statistics of "
|
||||
"JSON documents."
|
||||
<< std::endl;
|
||||
std::cerr << "Usage: " << argv[0] << " <jsonfile>" << std::endl;
|
||||
std::cerr << "Or " << argv[0] << " -v <jsonfile>" << std::endl;
|
||||
exit(1);
|
||||
}
|
||||
const char *filename = argv[optind];
|
||||
if (optind + 1 < argc) {
|
||||
cerr << "warning: ignoring everything after " << argv[optind + 1] << endl;
|
||||
std::cerr << "warning: ignoring everything after " << argv[optind + 1]
|
||||
<< std::endl;
|
||||
}
|
||||
std::string_view p;
|
||||
try {
|
||||
p = get_corpus(filename);
|
||||
} catch (const std::exception &e) { // caught by reference to base
|
||||
std::cout << "Could not load the file " << filename << std::endl;
|
||||
simdjson::padded_string p;
|
||||
auto error = simdjson::padded_string::load(filename).get(p);
|
||||
if (error) {
|
||||
std::cerr << "Could not load the file " << filename << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
|
||||
// Gigabyte: https://en.wikipedia.org/wiki/Gigabyte
|
||||
if (verbose) {
|
||||
std::cout << "Input has ";
|
||||
if (p.size() > 1024 * 1024)
|
||||
std::cout << p.size() / (1024 * 1024) << " MB ";
|
||||
else if (p.size() > 1024)
|
||||
std::cout << p.size() / 1024 << " KB ";
|
||||
if (p.size() > 1000 * 1000)
|
||||
std::cout << p.size() / (1000 * 1000) << " MB ";
|
||||
else if (p.size() > 1000)
|
||||
std::cout << p.size() / 1000 << " KB ";
|
||||
else
|
||||
std::cout << p.size() << " B ";
|
||||
std::cout << std::endl;
|
||||
}
|
||||
std::vector<int64_t> s1 = simdjson_computestats(p);
|
||||
std::vector<int64_t> s1 = simdjson_compute_stats(p);
|
||||
if (verbose) {
|
||||
printf("simdjson: ");
|
||||
print_vec(s1);
|
||||
}
|
||||
std::vector<int64_t> s2 = rapid_computestats(p);
|
||||
std::vector<int64_t> s2 = rapid_compute_stats(p);
|
||||
if (verbose) {
|
||||
printf("rapid: ");
|
||||
print_vec(s2);
|
||||
}
|
||||
std::vector<int64_t> s3 = sasjon_computestats(p);
|
||||
std::vector<int64_t> s3 = sasjon_compute_stats(p);
|
||||
if (verbose) {
|
||||
printf("sasjon: ");
|
||||
print_vec(s3);
|
||||
@@ -267,17 +367,40 @@ int main(int argc, char *argv[]) {
|
||||
assert(s1 == s3);
|
||||
size_t size = s1.size();
|
||||
|
||||
int repeat = 50;
|
||||
int volume = p.size();
|
||||
if(justdata) {
|
||||
printf("name cycles_per_byte cycles_per_byte_err gb_per_s gb_per_s_err \n");
|
||||
int repeat = 500;
|
||||
size_t volume = p.size();
|
||||
if (just_data) {
|
||||
printf(
|
||||
"name cycles_per_byte cycles_per_byte_err gb_per_s gb_per_s_err \n");
|
||||
}
|
||||
BEST_TIME("simdjson ", simdjson_computestats(p).size(), size, , repeat,
|
||||
volume, !justdata);
|
||||
|
||||
BEST_TIME("rapid ", rapid_computestats(p).size(), size, , repeat, volume,
|
||||
!justdata);
|
||||
BEST_TIME("sasjon ", sasjon_computestats(p).size(), size, , repeat, volume,
|
||||
!justdata);
|
||||
aligned_free((void*)p.data());
|
||||
BEST_TIME("simdjson ", simdjson_compute_stats(p).size(), size, , repeat,
|
||||
volume, !just_data);
|
||||
BEST_TIME("rapid ", rapid_compute_stats(p).size(), size, , repeat, volume,
|
||||
!just_data);
|
||||
BEST_TIME("sasjon ", sasjon_compute_stats(p).size(), size, , repeat, volume,
|
||||
!just_data);
|
||||
BEST_TIME("simdjson (just parse) ", simdjson_just_parse(p), simdjson::error_code::SUCCESS, , repeat,
|
||||
volume, !just_data);
|
||||
BEST_TIME("rapid (just parse) ", rapid_just_parse(p), false, , repeat,
|
||||
volume, !just_data);
|
||||
BEST_TIME("sasjon (just parse) ", sasjon_just_parse(p), false, , repeat,
|
||||
volume, !just_data);
|
||||
simdjson::dom::parser parser;
|
||||
simdjson::dom::element doc;
|
||||
error = parser.parse(p).get(doc);
|
||||
BEST_TIME("simdjson (just dom) ", simdjson_just_dom(doc).size(), size,
|
||||
, repeat, volume, !just_data);
|
||||
char *buffer = (char *)malloc(p.size() + 1);
|
||||
buffer[p.size()] = '\0';
|
||||
memcpy(buffer, p.data(), p.size());
|
||||
rapidjson::Document drapid;
|
||||
drapid.ParseInsitu<kParseValidateEncodingFlag>(buffer);
|
||||
BEST_TIME("rapid (just dom) ", rapid_just_dom(drapid).size(), size, , repeat,
|
||||
volume, !just_data);
|
||||
memcpy(buffer, p.data(), p.size());
|
||||
auto dsasjon = sajson::parse(sajson::dynamic_allocation(),
|
||||
sajson::mutable_string_view(p.size(), buffer));
|
||||
BEST_TIME("sasjon (just dom) ", sasjon_just_dom(dsasjon).size(), size, ,
|
||||
repeat, volume, !just_data);
|
||||
free(buffer);
|
||||
}
|
||||
|
||||
@@ -0,0 +1,152 @@
|
||||
#ifndef __EVENT_COUNTER_H
|
||||
#define __EVENT_COUNTER_H
|
||||
|
||||
#include <cassert>
|
||||
#include <cctype>
|
||||
#ifndef _MSC_VER
|
||||
#include <dirent.h>
|
||||
#endif
|
||||
#include <unistd.h>
|
||||
#include <cinttypes>
|
||||
|
||||
#include <cstdio>
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
|
||||
#include <algorithm>
|
||||
#include <chrono>
|
||||
#include <cstring>
|
||||
#include <fstream>
|
||||
#include <iomanip>
|
||||
#include <iostream>
|
||||
#include <map>
|
||||
#include <set>
|
||||
#include <sstream>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "linux-perf-events.h"
|
||||
#ifdef __linux__
|
||||
#include <libgen.h>
|
||||
#endif
|
||||
|
||||
#include "simdjson.h"
|
||||
|
||||
using std::string;
|
||||
using std::vector;
|
||||
using std::chrono::steady_clock;
|
||||
using std::chrono::time_point;
|
||||
using std::chrono::duration;
|
||||
|
||||
struct event_count {
|
||||
duration<double> elapsed;
|
||||
vector<unsigned long long> event_counts;
|
||||
event_count() : elapsed(0), event_counts{0,0,0,0,0} {}
|
||||
event_count(const duration<double> _elapsed, const vector<unsigned long long> _event_counts) : elapsed(_elapsed), event_counts(_event_counts) {}
|
||||
event_count(const event_count& other): elapsed(other.elapsed), event_counts(other.event_counts) { }
|
||||
|
||||
// The types of counters (so we can read the getter more easily)
|
||||
enum event_counter_types {
|
||||
CPU_CYCLES,
|
||||
INSTRUCTIONS,
|
||||
BRANCH_MISSES,
|
||||
CACHE_REFERENCES,
|
||||
CACHE_MISSES
|
||||
};
|
||||
|
||||
double elapsed_sec() const { return duration<double>(elapsed).count(); }
|
||||
double elapsed_ns() const { return duration<double, std::nano>(elapsed).count(); }
|
||||
double cycles() const { return static_cast<double>(event_counts[CPU_CYCLES]); }
|
||||
double instructions() const { return static_cast<double>(event_counts[INSTRUCTIONS]); }
|
||||
double branch_misses() const { return static_cast<double>(event_counts[BRANCH_MISSES]); }
|
||||
double cache_references() const { return static_cast<double>(event_counts[CACHE_REFERENCES]); }
|
||||
double cache_misses() const { return static_cast<double>(event_counts[CACHE_MISSES]); }
|
||||
|
||||
event_count& operator=(const event_count& other) {
|
||||
this->elapsed = other.elapsed;
|
||||
this->event_counts = other.event_counts;
|
||||
return *this;
|
||||
}
|
||||
event_count operator+(const event_count& other) const {
|
||||
return event_count(elapsed+other.elapsed, {
|
||||
event_counts[0]+other.event_counts[0],
|
||||
event_counts[1]+other.event_counts[1],
|
||||
event_counts[2]+other.event_counts[2],
|
||||
event_counts[3]+other.event_counts[3],
|
||||
event_counts[4]+other.event_counts[4],
|
||||
});
|
||||
}
|
||||
|
||||
void operator+=(const event_count& other) {
|
||||
*this = *this + other;
|
||||
}
|
||||
};
|
||||
|
||||
struct event_aggregate {
|
||||
int iterations = 0;
|
||||
event_count total{};
|
||||
event_count best{};
|
||||
event_count worst{};
|
||||
|
||||
event_aggregate() {}
|
||||
|
||||
void operator<<(const event_count& other) {
|
||||
if (iterations == 0 || other.elapsed < best.elapsed) {
|
||||
best = other;
|
||||
}
|
||||
if (iterations == 0 || other.elapsed > worst.elapsed) {
|
||||
worst = other;
|
||||
}
|
||||
iterations++;
|
||||
total += other;
|
||||
}
|
||||
|
||||
double elapsed_sec() const { return total.elapsed_sec() / iterations; }
|
||||
double elapsed_ns() const { return total.elapsed_ns() / iterations; }
|
||||
double cycles() const { return total.cycles() / iterations; }
|
||||
double instructions() const { return total.instructions() / iterations; }
|
||||
double branch_misses() const { return total.branch_misses() / iterations; }
|
||||
double cache_references() const { return total.cache_references() / iterations; }
|
||||
double cache_misses() const { return total.cache_misses() / iterations; }
|
||||
};
|
||||
|
||||
struct event_collector {
|
||||
event_count count{};
|
||||
time_point<steady_clock> start_clock{};
|
||||
|
||||
#if defined(__linux__)
|
||||
LinuxEvents<PERF_TYPE_HARDWARE> linux_events;
|
||||
event_collector(bool quiet = false) : linux_events(vector<int>{
|
||||
PERF_COUNT_HW_CPU_CYCLES,
|
||||
PERF_COUNT_HW_INSTRUCTIONS,
|
||||
PERF_COUNT_HW_BRANCH_MISSES,
|
||||
PERF_COUNT_HW_CACHE_REFERENCES,
|
||||
PERF_COUNT_HW_CACHE_MISSES
|
||||
}, quiet) {}
|
||||
bool has_events() {
|
||||
return linux_events.is_working();
|
||||
}
|
||||
#else
|
||||
event_collector(simdjson_unused bool _quiet = false) {}
|
||||
bool has_events() {
|
||||
return false;
|
||||
}
|
||||
#endif
|
||||
|
||||
simdjson_really_inline void start() {
|
||||
#if defined(__linux)
|
||||
linux_events.start();
|
||||
#endif
|
||||
start_clock = steady_clock::now();
|
||||
}
|
||||
simdjson_really_inline event_count& end() {
|
||||
time_point<steady_clock> end_clock = steady_clock::now();
|
||||
#if defined(__linux)
|
||||
linux_events.end(count.event_counts);
|
||||
#endif
|
||||
count.elapsed = end_clock - start_clock;
|
||||
return count;
|
||||
}
|
||||
};
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,57 @@
|
||||
|
||||
#include "simdjson.h"
|
||||
#include <chrono>
|
||||
#include <cstring>
|
||||
#include <iostream>
|
||||
|
||||
// Gigabyte: https://en.wikipedia.org/wiki/Gigabyte
|
||||
simdjson_never_inline
|
||||
double bench(std::string filename, simdjson::padded_string& p) {
|
||||
std::chrono::time_point<std::chrono::steady_clock> start_clock =
|
||||
std::chrono::steady_clock::now();
|
||||
simdjson::padded_string::load(filename).first.swap(p);
|
||||
std::chrono::time_point<std::chrono::steady_clock> end_clock =
|
||||
std::chrono::steady_clock::now();
|
||||
std::chrono::duration<double> elapsed = end_clock - start_clock;
|
||||
return (static_cast<double>(p.size()) / (1000000000.)) / elapsed.count();
|
||||
}
|
||||
|
||||
int main(int argc, char *argv[]) {
|
||||
int optind = 1;
|
||||
if (optind >= argc) {
|
||||
std::cerr << "Reads document as far as possible. " << std::endl;
|
||||
std::cerr << "Usage: " << argv[0] << " <jsonfile>" << std::endl;
|
||||
exit(1);
|
||||
}
|
||||
const char *filename = argv[optind];
|
||||
if (optind + 1 < argc) {
|
||||
std::cerr << "warning: ignoring everything after " << argv[optind + 1]
|
||||
<< std::endl;
|
||||
}
|
||||
simdjson::padded_string p;
|
||||
bench(filename, p);
|
||||
double meanval = 0;
|
||||
double maxval = 0;
|
||||
double minval = 10000;
|
||||
std::cout << "file size: "<< (static_cast<double>(p.size()) / (1000000000.)) << " GB" <<std::endl;
|
||||
size_t times = p.size() > 1000000000 ? 5 : 50;
|
||||
#if __cpp_exceptions
|
||||
try {
|
||||
#endif
|
||||
for(size_t i = 0; i < times; i++) {
|
||||
double tval = bench(filename, p);
|
||||
if(maxval < tval) maxval = tval;
|
||||
if(minval > tval) minval = tval;
|
||||
meanval += tval;
|
||||
}
|
||||
#if __cpp_exceptions
|
||||
} catch (const std::exception &) { // caught by reference to base
|
||||
std::cerr << "Could not load the file " << filename << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
#endif
|
||||
std::cout << "average speed: " << meanval / static_cast<double>(times) << " GB/s"<< std::endl;
|
||||
std::cout << "min speed : " << minval << " GB/s" << std::endl;
|
||||
std::cout << "max speed : " << maxval << " GB/s" << std::endl;
|
||||
return EXIT_SUCCESS;
|
||||
}
|
||||
@@ -0,0 +1,80 @@
|
||||
#pragma once
|
||||
|
||||
template<typename B, typename R> static void JsonBenchmark(benchmark::State &state, const simdjson::padded_string &json) {
|
||||
event_collector collector(true);
|
||||
event_aggregate events;
|
||||
|
||||
// Warmup and equality check (make sure the data is right!)
|
||||
B bench;
|
||||
if (!bench.Run(json)) { state.SkipWithError("warmup tweet reading failed"); return; }
|
||||
{
|
||||
R reference;
|
||||
if (!reference.Run(json)) { state.SkipWithError("reference tweet reading failed"); return; }
|
||||
if (bench.Result() != reference.Result()) { state.SkipWithError("results are not the same"); return; }
|
||||
}
|
||||
|
||||
// Run the benchmark
|
||||
for (simdjson_unused auto _ : state) {
|
||||
collector.start();
|
||||
|
||||
if (!bench.Run(json)) { state.SkipWithError("tweet reading failed"); return; }
|
||||
|
||||
events << collector.end();
|
||||
}
|
||||
|
||||
state.SetBytesProcessed(json.size() * state.iterations());
|
||||
state.SetItemsProcessed(bench.ItemCount() * state.iterations());
|
||||
state.counters["best_bytes_per_sec"] = benchmark::Counter(double(json.size()) / events.best.elapsed_sec());
|
||||
state.counters["best_items_per_sec"] = benchmark::Counter(double(bench.ItemCount()) / events.best.elapsed_sec());
|
||||
|
||||
state.counters["docs_per_sec"] = benchmark::Counter(1.0, benchmark::Counter::kIsIterationInvariantRate);
|
||||
state.counters["best_docs_per_sec"] = benchmark::Counter(1.0 / events.best.elapsed_sec());
|
||||
|
||||
if (collector.has_events()) {
|
||||
state.counters["instructions"] = events.instructions();
|
||||
state.counters["cycles"] = events.cycles();
|
||||
state.counters["branch_miss"] = events.branch_misses();
|
||||
state.counters["cache_miss"] = events.cache_misses();
|
||||
state.counters["cache_ref"] = events.cache_references();
|
||||
|
||||
state.counters["instructions_per_byte"] = events.instructions() / double(json.size());
|
||||
state.counters["instructions_per_cycle"] = events.instructions() / events.cycles();
|
||||
state.counters["cycles_per_byte"] = events.cycles() / double(json.size());
|
||||
state.counters["frequency"] = benchmark::Counter(events.cycles(), benchmark::Counter::kIsIterationInvariantRate);
|
||||
|
||||
state.counters["best_instructions"] = events.best.instructions();
|
||||
state.counters["best_cycles"] = events.best.cycles();
|
||||
state.counters["best_branch_miss"] = events.best.branch_misses();
|
||||
state.counters["best_cache_miss"] = events.best.cache_misses();
|
||||
state.counters["best_cache_ref"] = events.best.cache_references();
|
||||
|
||||
state.counters["best_instructions_per_byte"] = events.best.instructions() / double(json.size());
|
||||
state.counters["best_instructions_per_cycle"] = events.best.instructions() / events.best.cycles();
|
||||
state.counters["best_cycles_per_byte"] = events.best.cycles() / double(json.size());
|
||||
state.counters["best_frequency"] = events.best.cycles() / events.best.elapsed_sec();
|
||||
}
|
||||
state.counters["bytes"] = benchmark::Counter(double(json.size()));
|
||||
state.counters["items"] = benchmark::Counter(double(bench.ItemCount()));
|
||||
|
||||
// Build the label
|
||||
using namespace std;
|
||||
stringstream label;
|
||||
label << fixed << setprecision(2);
|
||||
label << "[best:";
|
||||
label << " throughput=" << setw(6) << (double(json.size()) / 1000000000.0 / events.best.elapsed_sec()) << " GB/s";
|
||||
label << " doc_throughput=" << setw(6) << uint64_t(1.0 / events.best.elapsed_sec()) << " docs/s";
|
||||
|
||||
if (collector.has_events()) {
|
||||
label << " instructions=" << setw(12) << uint64_t(events.best.instructions()) << setw(0);
|
||||
label << " cycles=" << setw(12) << uint64_t(events.best.cycles()) << setw(0);
|
||||
label << " branch_miss=" << setw(8) << uint64_t(events.best.branch_misses()) << setw(0);
|
||||
label << " cache_miss=" << setw(8) << uint64_t(events.best.cache_misses()) << setw(0);
|
||||
label << " cache_ref=" << setw(10) << uint64_t(events.best.cache_references()) << setw(0);
|
||||
}
|
||||
|
||||
label << " items=" << setw(10) << bench.ItemCount() << setw(0);
|
||||
label << " avg_time=" << setw(10) << uint64_t(events.elapsed_ns()) << setw(0) << " ns";
|
||||
label << "]";
|
||||
|
||||
state.SetLabel(label.str());
|
||||
}
|
||||
@@ -0,0 +1,69 @@
|
||||
#pragma once
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
|
||||
#include "kostya.h"
|
||||
|
||||
namespace kostya {
|
||||
|
||||
using namespace simdjson;
|
||||
|
||||
class Dom {
|
||||
public:
|
||||
simdjson_really_inline bool Run(const padded_string &json);
|
||||
|
||||
simdjson_really_inline const std::vector<my_point> &Result() { return container; }
|
||||
simdjson_really_inline size_t ItemCount() { return container.size(); }
|
||||
|
||||
private:
|
||||
dom::parser parser{};
|
||||
std::vector<my_point> container{};
|
||||
};
|
||||
|
||||
simdjson_really_inline bool Dom::Run(const padded_string &json) {
|
||||
container.clear();
|
||||
|
||||
for (auto point : parser.parse(json)["coordinates"]) {
|
||||
container.emplace_back(my_point{point["x"], point["y"], point["z"]});
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
BENCHMARK_TEMPLATE(Kostya, Dom);
|
||||
|
||||
namespace sum {
|
||||
|
||||
class Dom {
|
||||
public:
|
||||
simdjson_really_inline bool Run(const padded_string &json);
|
||||
|
||||
simdjson_really_inline my_point &Result() { return sum; }
|
||||
simdjson_really_inline size_t ItemCount() { return count; }
|
||||
|
||||
private:
|
||||
dom::parser parser{};
|
||||
my_point sum{};
|
||||
size_t count{};
|
||||
};
|
||||
|
||||
simdjson_really_inline bool Dom::Run(const padded_string &json) {
|
||||
sum = { 0, 0, 0 };
|
||||
count = 0;
|
||||
|
||||
for (auto coord : parser.parse(json)["coordinates"]) {
|
||||
sum.x += double(coord["x"]);
|
||||
sum.y += double(coord["y"]);
|
||||
sum.z += double(coord["z"]);
|
||||
count++;
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
BENCHMARK_TEMPLATE(KostyaSum, Dom);
|
||||
|
||||
} // namespace sum
|
||||
} // namespace kostya
|
||||
|
||||
#endif // SIMDJSON_EXCEPTIONS
|
||||
@@ -0,0 +1,96 @@
|
||||
#pragma once
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
|
||||
#include "kostya.h"
|
||||
|
||||
namespace kostya {
|
||||
|
||||
using namespace simdjson;
|
||||
using namespace simdjson::builtin;
|
||||
|
||||
class Iter {
|
||||
public:
|
||||
simdjson_really_inline bool Run(const padded_string &json);
|
||||
|
||||
simdjson_really_inline const std::vector<my_point> &Result() { return container; }
|
||||
simdjson_really_inline size_t ItemCount() { return container.size(); }
|
||||
|
||||
private:
|
||||
ondemand::parser parser{};
|
||||
std::vector<my_point> container{};
|
||||
|
||||
simdjson_really_inline simdjson_result<double> first_double(ondemand::json_iterator &iter, const char *key) {
|
||||
if (!iter.start_object() || ondemand::raw_json_string(iter.field_key()) != key || iter.field_value()) { throw "Invalid field"; }
|
||||
return iter.consume_double();
|
||||
}
|
||||
|
||||
simdjson_really_inline simdjson_result<double> next_double(ondemand::json_iterator &iter, const char *key) {
|
||||
if (!iter.has_next_field() || ondemand::raw_json_string(iter.field_key()) != key || iter.field_value()) { throw "Invalid field"; }
|
||||
return iter.consume_double();
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
simdjson_really_inline bool Iter::Run(const padded_string &json) {
|
||||
container.clear();
|
||||
|
||||
using std::cerr;
|
||||
using std::endl;
|
||||
auto iter = parser.iterate_raw(json).value();
|
||||
if (!iter.start_object() || !iter.find_field_raw("coordinates")) { cerr << "find coordinates field failed" << endl; return false; }
|
||||
if (iter.start_array()) {
|
||||
do {
|
||||
container.emplace_back(my_point{first_double(iter, "x"), next_double(iter, "y"), next_double(iter, "z")});
|
||||
if (iter.skip_container()) { return false; } // Skip the rest of the coordinates object
|
||||
} while (iter.has_next_element());
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
BENCHMARK_TEMPLATE(Kostya, Iter);
|
||||
|
||||
|
||||
namespace sum {
|
||||
|
||||
class Iter {
|
||||
public:
|
||||
simdjson_really_inline bool Run(const padded_string &json);
|
||||
|
||||
simdjson_really_inline my_point &Result() { return sum; }
|
||||
simdjson_really_inline size_t ItemCount() { return count; }
|
||||
|
||||
private:
|
||||
ondemand::parser parser{};
|
||||
my_point sum{};
|
||||
size_t count{};
|
||||
};
|
||||
|
||||
simdjson_really_inline bool Iter::Run(const padded_string &json) {
|
||||
sum = {0,0,0};
|
||||
count = 0;
|
||||
|
||||
auto iter = parser.iterate_raw(json).value();
|
||||
if (!iter.start_object() || !iter.find_field_raw("coordinates")) { return false; }
|
||||
if (!iter.start_array()) { return false; }
|
||||
do {
|
||||
if (!iter.start_object() || !iter.find_field_raw("x")) { return false; }
|
||||
sum.x += iter.consume_double();
|
||||
if (!iter.has_next_field() || !iter.find_field_raw("y")) { return false; }
|
||||
sum.y += iter.consume_double();
|
||||
if (!iter.has_next_field() || !iter.find_field_raw("z")) { return false; }
|
||||
sum.z += iter.consume_double();
|
||||
if (iter.skip_container()) { return false; } // Skip the rest of the coordinates object
|
||||
count++;
|
||||
} while (iter.has_next_element());
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
BENCHMARK_TEMPLATE(KostyaSum, Iter);
|
||||
|
||||
} // namespace sum
|
||||
} // namespace kostya
|
||||
|
||||
#endif // SIMDJSON_EXCEPTIONS
|
||||
@@ -0,0 +1,95 @@
|
||||
#pragma once
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
|
||||
//
|
||||
// Interface
|
||||
//
|
||||
|
||||
namespace kostya {
|
||||
template<typename T> static void Kostya(benchmark::State &state);
|
||||
namespace sum {
|
||||
template<typename T> static void KostyaSum(benchmark::State &state);
|
||||
}
|
||||
|
||||
using namespace simdjson;
|
||||
|
||||
static void append_coordinate(std::default_random_engine &e, std::uniform_real_distribution<> &dis, std::stringstream &myss) {
|
||||
using std::endl;
|
||||
myss << R"( {)" << endl;
|
||||
myss << R"( "x": )" << dis(e) << "," << endl;
|
||||
myss << R"( "y": )" << dis(e) << "," << endl;
|
||||
myss << R"( "z": )" << dis(e) << "," << endl;
|
||||
myss << R"( "name": ")" << char('a'+dis(e)*25) << char('a'+dis(e)*25) << char('a'+dis(e)*25) << char('a'+dis(e)*25) << char('a'+dis(e)*25) << char('a'+dis(e)*25) << " " << int(dis(e)*10000) << "\"," << endl;
|
||||
myss << R"( "opts": {)" << endl;
|
||||
myss << R"( "1": [)" << endl;
|
||||
myss << R"( 1,)" << endl;
|
||||
myss << R"( true)" << endl;
|
||||
myss << R"( ])" << endl;
|
||||
myss << R"( })" << endl;
|
||||
myss << R"( })";
|
||||
}
|
||||
|
||||
static std::string build_json_array(size_t N) {
|
||||
using namespace std;
|
||||
default_random_engine e;
|
||||
uniform_real_distribution<> dis(0, 1);
|
||||
stringstream myss;
|
||||
myss << R"({)" << endl;
|
||||
myss << R"( "coordinates": [)" << endl;
|
||||
for (size_t i=1; i<N; i++) {
|
||||
append_coordinate(e, dis, myss); myss << "," << endl;
|
||||
}
|
||||
append_coordinate(e, dis, myss); myss << endl;
|
||||
myss << R"( ],)" << endl;
|
||||
myss << R"( "info": "some info")" << endl;
|
||||
myss << R"(})" << endl;
|
||||
string answer = myss.str();
|
||||
cout << "Creating a source file spanning " << (answer.size() + 512) / 1024 << " KB " << endl;
|
||||
return answer;
|
||||
}
|
||||
|
||||
static const padded_string &get_built_json_array() {
|
||||
static padded_string json = build_json_array(524288);
|
||||
return json;
|
||||
}
|
||||
|
||||
struct my_point {
|
||||
double x;
|
||||
double y;
|
||||
double z;
|
||||
simdjson_really_inline bool operator==(const my_point &other) const {
|
||||
return x == other.x && y == other.y && z == other.z;
|
||||
}
|
||||
simdjson_really_inline bool operator!=(const my_point &other) const { return !(*this == other); }
|
||||
};
|
||||
|
||||
simdjson_unused static std::ostream &operator<<(std::ostream &o, const my_point &p) {
|
||||
return o << p.x << "," << p.y << "," << p.z << std::endl;
|
||||
}
|
||||
|
||||
} // namespace kostya
|
||||
|
||||
//
|
||||
// Implementation
|
||||
//
|
||||
#include <vector>
|
||||
#include "event_counter.h"
|
||||
#include "dom.h"
|
||||
#include "json_benchmark.h"
|
||||
|
||||
namespace kostya {
|
||||
|
||||
template<typename T> static void Kostya(benchmark::State &state) {
|
||||
JsonBenchmark<T, Dom>(state, get_built_json_array());
|
||||
}
|
||||
|
||||
namespace sum {
|
||||
template<typename T> static void KostyaSum(benchmark::State &state) {
|
||||
JsonBenchmark<T, Dom>(state, get_built_json_array());
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace kostya
|
||||
|
||||
#endif // SIMDJSON_EXCEPTIONS
|
||||
@@ -0,0 +1,74 @@
|
||||
#pragma once
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
|
||||
#include "kostya.h"
|
||||
|
||||
namespace kostya {
|
||||
|
||||
using namespace simdjson;
|
||||
using namespace simdjson::builtin;
|
||||
|
||||
class OnDemand {
|
||||
public:
|
||||
simdjson_really_inline bool Run(const padded_string &json);
|
||||
simdjson_really_inline const std::vector<my_point> &Result() { return container; }
|
||||
simdjson_really_inline size_t ItemCount() { return container.size(); }
|
||||
|
||||
private:
|
||||
ondemand::parser parser{};
|
||||
std::vector<my_point> container{};
|
||||
};
|
||||
|
||||
simdjson_really_inline bool OnDemand::Run(const padded_string &json) {
|
||||
container.clear();
|
||||
|
||||
using std::cout;
|
||||
using std::endl;
|
||||
|
||||
auto doc = parser.iterate(json);
|
||||
for (ondemand::object coord : doc["coordinates"]) {
|
||||
container.emplace_back(my_point{coord["x"], coord["y"], coord["z"]});
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
BENCHMARK_TEMPLATE(Kostya, OnDemand);
|
||||
|
||||
|
||||
namespace sum {
|
||||
|
||||
class OnDemand {
|
||||
public:
|
||||
simdjson_really_inline bool Run(const padded_string &json);
|
||||
simdjson_really_inline my_point &Result() { return sum; }
|
||||
simdjson_really_inline size_t ItemCount() { return count; }
|
||||
|
||||
private:
|
||||
ondemand::parser parser{};
|
||||
my_point sum{};
|
||||
size_t count{};
|
||||
};
|
||||
|
||||
simdjson_really_inline bool OnDemand::Run(const padded_string &json) {
|
||||
sum = {0,0,0};
|
||||
count = 0;
|
||||
|
||||
auto doc = parser.iterate(json);
|
||||
for (ondemand::object coord : doc["coordinates"]) {
|
||||
sum.x += double(coord["x"]);
|
||||
sum.y += double(coord["y"]);
|
||||
sum.z += double(coord["z"]);
|
||||
count++;
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
BENCHMARK_TEMPLATE(KostyaSum, OnDemand);
|
||||
|
||||
} // namespace sum
|
||||
} // namespace kostya
|
||||
|
||||
#endif // SIMDJSON_EXCEPTIONS
|
||||
@@ -0,0 +1,69 @@
|
||||
#pragma once
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
|
||||
#include "largerandom.h"
|
||||
|
||||
namespace largerandom {
|
||||
|
||||
using namespace simdjson;
|
||||
|
||||
class Dom {
|
||||
public:
|
||||
simdjson_really_inline bool Run(const padded_string &json);
|
||||
|
||||
simdjson_really_inline const std::vector<my_point> &Result() { return container; }
|
||||
simdjson_really_inline size_t ItemCount() { return container.size(); }
|
||||
|
||||
private:
|
||||
dom::parser parser{};
|
||||
std::vector<my_point> container{};
|
||||
};
|
||||
|
||||
simdjson_really_inline bool Dom::Run(const padded_string &json) {
|
||||
container.clear();
|
||||
|
||||
for (auto point : parser.parse(json)) {
|
||||
container.emplace_back(my_point{point["x"], point["y"], point["z"]});
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
BENCHMARK_TEMPLATE(LargeRandom, Dom);
|
||||
|
||||
namespace sum {
|
||||
|
||||
class Dom {
|
||||
public:
|
||||
simdjson_really_inline bool Run(const padded_string &json);
|
||||
|
||||
simdjson_really_inline my_point &Result() { return sum; }
|
||||
simdjson_really_inline size_t ItemCount() { return count; }
|
||||
|
||||
private:
|
||||
dom::parser parser{};
|
||||
my_point sum{};
|
||||
size_t count{};
|
||||
};
|
||||
|
||||
simdjson_really_inline bool Dom::Run(const padded_string &json) {
|
||||
sum = { 0, 0, 0 };
|
||||
count = 0;
|
||||
|
||||
for (auto coord : parser.parse(json)) {
|
||||
sum.x += double(coord["x"]);
|
||||
sum.y += double(coord["y"]);
|
||||
sum.z += double(coord["z"]);
|
||||
count++;
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
BENCHMARK_TEMPLATE(LargeRandomSum, Dom);
|
||||
|
||||
} // namespace sum
|
||||
} // namespace largerandom
|
||||
|
||||
#endif // SIMDJSON_EXCEPTIONS
|
||||
@@ -0,0 +1,92 @@
|
||||
#pragma once
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
|
||||
#include "largerandom.h"
|
||||
|
||||
namespace largerandom {
|
||||
|
||||
using namespace simdjson;
|
||||
using namespace simdjson::builtin;
|
||||
|
||||
class Iter {
|
||||
public:
|
||||
simdjson_really_inline bool Run(const padded_string &json);
|
||||
|
||||
simdjson_really_inline const std::vector<my_point> &Result() { return container; }
|
||||
simdjson_really_inline size_t ItemCount() { return container.size(); }
|
||||
|
||||
private:
|
||||
ondemand::parser parser{};
|
||||
std::vector<my_point> container{};
|
||||
|
||||
simdjson_really_inline double first_double(ondemand::json_iterator &iter) {
|
||||
if (iter.start_object().error() || iter.field_key().error() || iter.field_value()) { throw "Invalid field"; }
|
||||
return iter.consume_double();
|
||||
}
|
||||
|
||||
simdjson_really_inline double next_double(ondemand::json_iterator &iter) {
|
||||
if (!iter.has_next_field() || iter.field_key().error() || iter.field_value()) { throw "Invalid field"; }
|
||||
return iter.consume_double();
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
simdjson_really_inline bool Iter::Run(const padded_string &json) {
|
||||
container.clear();
|
||||
|
||||
auto iter = parser.iterate_raw(json).value();
|
||||
if (iter.start_array()) {
|
||||
do {
|
||||
container.emplace_back(my_point{first_double(iter), next_double(iter), next_double(iter)});
|
||||
if (iter.has_next_field()) { throw "Too many fields"; }
|
||||
} while (iter.has_next_element());
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
BENCHMARK_TEMPLATE(LargeRandom, Iter);
|
||||
|
||||
|
||||
namespace sum {
|
||||
|
||||
class Iter {
|
||||
public:
|
||||
simdjson_really_inline bool Run(const padded_string &json);
|
||||
|
||||
simdjson_really_inline my_point &Result() { return sum; }
|
||||
simdjson_really_inline size_t ItemCount() { return count; }
|
||||
|
||||
private:
|
||||
ondemand::parser parser{};
|
||||
my_point sum{};
|
||||
size_t count{};
|
||||
};
|
||||
|
||||
simdjson_really_inline bool Iter::Run(const padded_string &json) {
|
||||
sum = {0,0,0};
|
||||
count = 0;
|
||||
|
||||
auto iter = parser.iterate_raw(json).value();
|
||||
if (!iter.start_array()) { return false; }
|
||||
do {
|
||||
if (!iter.start_object() || iter.field_key().value() != "x" || iter.field_value()) { return false; }
|
||||
sum.x += iter.consume_double();
|
||||
if (!iter.has_next_field() || iter.field_key().value() != "y" || iter.field_value()) { return false; }
|
||||
sum.y += iter.consume_double();
|
||||
if (!iter.has_next_field() || iter.field_key().value() != "z" || iter.field_value()) { return false; }
|
||||
sum.z += iter.consume_double();
|
||||
if (*iter.advance() != '}') { return false; }
|
||||
count++;
|
||||
} while (iter.has_next_element());
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
BENCHMARK_TEMPLATE(LargeRandomSum, Iter);
|
||||
|
||||
} // namespace sum
|
||||
} // namespace largerandom
|
||||
|
||||
#endif // SIMDJSON_EXCEPTIONS
|
||||
@@ -0,0 +1,80 @@
|
||||
#pragma once
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
|
||||
//
|
||||
// Interface
|
||||
//
|
||||
|
||||
namespace largerandom {
|
||||
template<typename T> static void LargeRandom(benchmark::State &state);
|
||||
namespace sum {
|
||||
template<typename T> static void LargeRandomSum(benchmark::State &state);
|
||||
}
|
||||
|
||||
using namespace simdjson;
|
||||
|
||||
static std::string build_json_array(size_t N) {
|
||||
std::default_random_engine e;
|
||||
std::uniform_real_distribution<> dis(0, 1);
|
||||
std::stringstream myss;
|
||||
myss << "[" << std::endl;
|
||||
if(N > 0) {
|
||||
myss << "{ \"x\":" << dis(e) << ", \"y\":" << dis(e) << ", \"z\":" << dis(e) << "}" << std::endl;
|
||||
}
|
||||
for(size_t i = 1; i < N; i++) {
|
||||
myss << "," << std::endl;
|
||||
myss << "{ \"x\":" << dis(e) << ", \"y\":" << dis(e) << ", \"z\":" << dis(e) << "}";
|
||||
}
|
||||
myss << std::endl;
|
||||
myss << "]" << std::endl;
|
||||
std::string answer = myss.str();
|
||||
std::cout << "Creating a source file spanning " << (answer.size() + 512) / 1024 << " KB " << std::endl;
|
||||
return answer;
|
||||
}
|
||||
|
||||
static const padded_string &get_built_json_array() {
|
||||
static padded_string json = build_json_array(1000000);
|
||||
return json;
|
||||
}
|
||||
|
||||
struct my_point {
|
||||
double x;
|
||||
double y;
|
||||
double z;
|
||||
simdjson_really_inline bool operator==(const my_point &other) const {
|
||||
return x == other.x && y == other.y && z == other.z;
|
||||
}
|
||||
simdjson_really_inline bool operator!=(const my_point &other) const { return !(*this == other); }
|
||||
};
|
||||
|
||||
simdjson_unused static std::ostream &operator<<(std::ostream &o, const my_point &p) {
|
||||
return o << p.x << "," << p.y << "," << p.z << std::endl;
|
||||
}
|
||||
|
||||
} // namespace largerandom
|
||||
|
||||
//
|
||||
// Implementation
|
||||
//
|
||||
#include <vector>
|
||||
#include "event_counter.h"
|
||||
#include "dom.h"
|
||||
#include "json_benchmark.h"
|
||||
|
||||
namespace largerandom {
|
||||
|
||||
template<typename T> static void LargeRandom(benchmark::State &state) {
|
||||
JsonBenchmark<T, Dom>(state, get_built_json_array());
|
||||
}
|
||||
|
||||
namespace sum {
|
||||
|
||||
template<typename T> static void LargeRandomSum(benchmark::State &state) {
|
||||
JsonBenchmark<T, Dom>(state, get_built_json_array());
|
||||
}
|
||||
|
||||
}
|
||||
} // namespace largerandom
|
||||
|
||||
#endif // SIMDJSON_EXCEPTIONS
|
||||
@@ -0,0 +1,71 @@
|
||||
#pragma once
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
|
||||
#include "largerandom.h"
|
||||
|
||||
namespace largerandom {
|
||||
|
||||
using namespace simdjson;
|
||||
using namespace simdjson::builtin;
|
||||
|
||||
class OnDemand {
|
||||
public:
|
||||
simdjson_really_inline bool Run(const padded_string &json);
|
||||
simdjson_really_inline const std::vector<my_point> &Result() { return container; }
|
||||
simdjson_really_inline size_t ItemCount() { return container.size(); }
|
||||
|
||||
private:
|
||||
ondemand::parser parser{};
|
||||
std::vector<my_point> container{};
|
||||
};
|
||||
|
||||
simdjson_really_inline bool OnDemand::Run(const padded_string &json) {
|
||||
container.clear();
|
||||
|
||||
auto doc = parser.iterate(json);
|
||||
for (ondemand::object coord : doc) {
|
||||
container.emplace_back(my_point{coord["x"], coord["y"], coord["z"]});
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
BENCHMARK_TEMPLATE(LargeRandom, OnDemand);
|
||||
|
||||
|
||||
namespace sum {
|
||||
|
||||
class OnDemand {
|
||||
public:
|
||||
simdjson_really_inline bool Run(const padded_string &json);
|
||||
simdjson_really_inline my_point &Result() { return sum; }
|
||||
simdjson_really_inline size_t ItemCount() { return count; }
|
||||
|
||||
private:
|
||||
ondemand::parser parser{};
|
||||
my_point sum{};
|
||||
size_t count{};
|
||||
};
|
||||
|
||||
simdjson_really_inline bool OnDemand::Run(const padded_string &json) {
|
||||
sum = {0,0,0};
|
||||
count = 0;
|
||||
|
||||
auto doc = parser.iterate(json);
|
||||
for (ondemand::object coord : doc.get_array()) {
|
||||
sum.x += double(coord["x"]);
|
||||
sum.y += double(coord["y"]);
|
||||
sum.z += double(coord["z"]);
|
||||
count++;
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
BENCHMARK_TEMPLATE(LargeRandomSum, OnDemand);
|
||||
|
||||
} // namespace sum
|
||||
} // namespace largerandom
|
||||
|
||||
#endif // SIMDJSON_EXCEPTIONS
|
||||
@@ -0,0 +1,121 @@
|
||||
#pragma once
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
|
||||
#include "largerandom.h"
|
||||
|
||||
namespace largerandom {
|
||||
|
||||
using namespace simdjson;
|
||||
using namespace simdjson::builtin;
|
||||
using namespace simdjson::builtin::stage2;
|
||||
|
||||
class Sax {
|
||||
public:
|
||||
simdjson_really_inline bool Run(const padded_string &json) noexcept;
|
||||
|
||||
simdjson_really_inline const std::vector<my_point> &Result() { return container; }
|
||||
simdjson_really_inline size_t ItemCount() { return container.size(); }
|
||||
|
||||
private:
|
||||
simdjson_really_inline error_code RunNoExcept(const padded_string &json) noexcept;
|
||||
error_code Allocate(size_t new_capacity);
|
||||
std::unique_ptr<uint8_t[]> string_buf{};
|
||||
size_t capacity{};
|
||||
dom_parser_implementation dom_parser{};
|
||||
std::vector<my_point> container{};
|
||||
};
|
||||
|
||||
struct sax_point_reader_visitor {
|
||||
public:
|
||||
std::vector<my_point> &points;
|
||||
enum {GOT_X=0, GOT_Y=1, GOT_Z=2, GOT_SOMETHING_ELSE=4};
|
||||
size_t idx{GOT_SOMETHING_ELSE};
|
||||
double buffer[3]={};
|
||||
|
||||
explicit sax_point_reader_visitor(std::vector<my_point> &_points) : points(_points) {}
|
||||
|
||||
simdjson_really_inline error_code visit_object_start(json_iterator &) {
|
||||
idx = 0;
|
||||
return SUCCESS;
|
||||
}
|
||||
simdjson_really_inline error_code visit_primitive(json_iterator &, const uint8_t *value) {
|
||||
if(idx == GOT_SOMETHING_ELSE) { return simdjson::SUCCESS; }
|
||||
return numberparsing::parse_double(value).get(buffer[idx]);
|
||||
}
|
||||
simdjson_really_inline error_code visit_object_end(json_iterator &) {
|
||||
points.emplace_back(my_point{buffer[0], buffer[1], buffer[2]});
|
||||
return SUCCESS;
|
||||
}
|
||||
|
||||
simdjson_really_inline error_code visit_document_start(json_iterator &) { return SUCCESS; }
|
||||
simdjson_really_inline error_code visit_key(json_iterator &, const uint8_t * key) {
|
||||
switch(key[1]) {
|
||||
// Technically, we should check the other characters
|
||||
// in the key, but we are cheating to go as fast
|
||||
// as possible.
|
||||
case 'x':
|
||||
idx = GOT_X;
|
||||
break;
|
||||
case 'y':
|
||||
idx = GOT_Y;
|
||||
break;
|
||||
case 'z':
|
||||
idx = GOT_Z;
|
||||
break;
|
||||
default:
|
||||
idx = GOT_SOMETHING_ELSE;
|
||||
}
|
||||
return SUCCESS;
|
||||
}
|
||||
simdjson_really_inline error_code visit_array_start(json_iterator &) { return SUCCESS; }
|
||||
simdjson_really_inline error_code visit_array_end(json_iterator &) { return SUCCESS; }
|
||||
simdjson_really_inline error_code visit_document_end(json_iterator &) { return SUCCESS; }
|
||||
simdjson_really_inline error_code visit_empty_array(json_iterator &) { return SUCCESS; }
|
||||
simdjson_really_inline error_code visit_empty_object(json_iterator &) { return SUCCESS; }
|
||||
simdjson_really_inline error_code visit_root_primitive(json_iterator &, const uint8_t *) { return SUCCESS; }
|
||||
simdjson_really_inline error_code increment_count(json_iterator &) { return SUCCESS; }
|
||||
};
|
||||
|
||||
// NOTE: this assumes the dom_parser is already allocated
|
||||
bool Sax::Run(const padded_string &json) noexcept {
|
||||
auto error = RunNoExcept(json);
|
||||
if (error) { std::cerr << error << std::endl; return false; }
|
||||
return true;
|
||||
}
|
||||
|
||||
error_code Sax::RunNoExcept(const padded_string &json) noexcept {
|
||||
container.clear();
|
||||
|
||||
// Allocate capacity if needed
|
||||
if (capacity < json.size()) {
|
||||
SIMDJSON_TRY( Allocate(json.size()) );
|
||||
}
|
||||
|
||||
// Run stage 1 first.
|
||||
SIMDJSON_TRY( dom_parser.stage1(json.u8data(), json.size(), false) );
|
||||
|
||||
// Then walk the document, parsing the tweets as we go
|
||||
json_iterator iter(dom_parser, 0);
|
||||
sax_point_reader_visitor visitor(container);
|
||||
SIMDJSON_TRY( iter.walk_document<false>(visitor) );
|
||||
return SUCCESS;
|
||||
}
|
||||
|
||||
error_code Sax::Allocate(size_t new_capacity) {
|
||||
// string_capacity copied from document::allocate
|
||||
size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * new_capacity / 3 + SIMDJSON_PADDING, 64);
|
||||
string_buf.reset(new (std::nothrow) uint8_t[string_capacity]);
|
||||
if (auto error = dom_parser.set_capacity(new_capacity)) { return error; }
|
||||
if (capacity == 0) { // set max depth the first time only
|
||||
if (auto error = dom_parser.set_max_depth(DEFAULT_MAX_DEPTH)) { return error; }
|
||||
}
|
||||
capacity = new_capacity;
|
||||
return SUCCESS;
|
||||
}
|
||||
|
||||
BENCHMARK_TEMPLATE(LargeRandom, Sax);
|
||||
|
||||
} // namespace largerandom
|
||||
|
||||
#endif // SIMDJSON_EXCEPTIONS
|
||||
@@ -1,29 +1,39 @@
|
||||
// https://github.com/WojciechMula/toys/blob/master/000helpers/linux-perf-events.h
|
||||
#pragma once
|
||||
#ifdef __linux__
|
||||
|
||||
#ifdef __has_include
|
||||
#if __has_include(<asm/unistd.h>)
|
||||
#include <asm/unistd.h> // for __NR_perf_event_open
|
||||
#else
|
||||
#warning "Header asm/unistd.h cannot be found though it is a linux system. Are linux headers missing?"
|
||||
#endif
|
||||
#else // no __has_include
|
||||
// Please insure that linux headers have been installed.
|
||||
#include <asm/unistd.h> // for __NR_perf_event_open
|
||||
#endif
|
||||
#include <linux/perf_event.h> // for perf event constants
|
||||
#include <sys/ioctl.h> // for ioctl
|
||||
#include <unistd.h> // for syscall
|
||||
|
||||
#include <cerrno> // for errno
|
||||
#include <cstring> // for memset
|
||||
#include <cstring> // for std::memset
|
||||
#include <stdexcept>
|
||||
|
||||
#include <iostream>
|
||||
#include <vector>
|
||||
|
||||
|
||||
template <int TYPE = PERF_TYPE_HARDWARE> class LinuxEvents {
|
||||
int fd;
|
||||
bool working;
|
||||
perf_event_attr attribs;
|
||||
int num_events;
|
||||
std::vector<uint64_t> temp_result_vec;
|
||||
std::vector<uint64_t> ids;
|
||||
bool working;
|
||||
perf_event_attr attribs{};
|
||||
size_t num_events{};
|
||||
std::vector<uint64_t> temp_result_vec{};
|
||||
std::vector<uint64_t> ids{};
|
||||
bool quiet;
|
||||
|
||||
public:
|
||||
explicit LinuxEvents(std::vector<int> config_vec) : fd(0), working(true) {
|
||||
memset(&attribs, 0, sizeof(attribs));
|
||||
explicit LinuxEvents(std::vector<int> config_vec, bool _quiet=false) : fd(0), working(true), quiet{_quiet} {
|
||||
std::memset(&attribs, 0, sizeof(attribs));
|
||||
attribs.type = TYPE;
|
||||
attribs.size = sizeof(attribs);
|
||||
attribs.disabled = 1;
|
||||
@@ -38,10 +48,11 @@ public:
|
||||
|
||||
int group = -1; // no group
|
||||
num_events = config_vec.size();
|
||||
ids.resize(config_vec.size());
|
||||
uint32_t i = 0;
|
||||
for (auto config : config_vec) {
|
||||
attribs.config = config;
|
||||
fd = syscall(__NR_perf_event_open, &attribs, pid, cpu, group, flags);
|
||||
fd = static_cast<int>(syscall(__NR_perf_event_open, &attribs, pid, cpu, group, flags));
|
||||
if (fd == -1) {
|
||||
report_error("perf_event_open");
|
||||
}
|
||||
@@ -54,25 +65,29 @@ public:
|
||||
temp_result_vec.resize(num_events * 2 + 1);
|
||||
}
|
||||
|
||||
~LinuxEvents() { close(fd); }
|
||||
~LinuxEvents() { if (fd != -1) { close(fd); } }
|
||||
|
||||
inline void start() {
|
||||
if (ioctl(fd, PERF_EVENT_IOC_RESET, PERF_IOC_FLAG_GROUP) == -1) {
|
||||
report_error("ioctl(PERF_EVENT_IOC_RESET)");
|
||||
}
|
||||
if (fd != -1) {
|
||||
if (ioctl(fd, PERF_EVENT_IOC_RESET, PERF_IOC_FLAG_GROUP) == -1) {
|
||||
report_error("ioctl(PERF_EVENT_IOC_RESET)");
|
||||
}
|
||||
|
||||
if (ioctl(fd, PERF_EVENT_IOC_ENABLE, PERF_IOC_FLAG_GROUP) == -1) {
|
||||
report_error("ioctl(PERF_EVENT_IOC_ENABLE)");
|
||||
if (ioctl(fd, PERF_EVENT_IOC_ENABLE, PERF_IOC_FLAG_GROUP) == -1) {
|
||||
report_error("ioctl(PERF_EVENT_IOC_ENABLE)");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
inline void end(std::vector<unsigned long long> &results) {
|
||||
if (ioctl(fd, PERF_EVENT_IOC_DISABLE, PERF_IOC_FLAG_GROUP) == -1) {
|
||||
report_error("ioctl(PERF_EVENT_IOC_DISABLE)");
|
||||
}
|
||||
if (fd != -1) {
|
||||
if (ioctl(fd, PERF_EVENT_IOC_DISABLE, PERF_IOC_FLAG_GROUP) == -1) {
|
||||
report_error("ioctl(PERF_EVENT_IOC_DISABLE)");
|
||||
}
|
||||
|
||||
if (read(fd, &temp_result_vec[0], temp_result_vec.size() * 8) == -1) {
|
||||
report_error("read");
|
||||
if (read(fd, temp_result_vec.data(), temp_result_vec.size() * 8) == -1) {
|
||||
report_error("read");
|
||||
}
|
||||
}
|
||||
// our actual results are in slots 1,3,5, ... of this structure
|
||||
// we really should be checking our ids obtained earlier to be safe
|
||||
@@ -81,10 +96,18 @@ public:
|
||||
}
|
||||
}
|
||||
|
||||
bool is_working() {
|
||||
return working;
|
||||
}
|
||||
|
||||
private:
|
||||
void report_error(const std::string &context) {
|
||||
if(working) std::cerr << (context + ": " + std::string(strerror(errno))) << std::endl;
|
||||
working = false;
|
||||
if (!quiet) {
|
||||
if (working) {
|
||||
std::cerr << (context + ": " + std::string(strerror(errno))) << std::endl;
|
||||
}
|
||||
}
|
||||
working = false;
|
||||
}
|
||||
};
|
||||
#endif
|
||||
|
||||
@@ -1,10 +1,10 @@
|
||||
#include <unistd.h>
|
||||
#include <iostream>
|
||||
#include <unistd.h>
|
||||
|
||||
#include "benchmark.h"
|
||||
#include "simdjson/jsonioutil.h"
|
||||
#include "simdjson/jsonminifier.h"
|
||||
#include "simdjson/jsonparser.h"
|
||||
#include "simdjson.h"
|
||||
|
||||
SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
|
||||
|
||||
// #define RAPIDJSON_SSE2 // bad
|
||||
// #define RAPIDJSON_SSE42 // bad
|
||||
@@ -14,11 +14,12 @@
|
||||
#include "rapidjson/writer.h"
|
||||
#include "sajson.h"
|
||||
|
||||
SIMDJSON_POP_DISABLE_WARNINGS
|
||||
|
||||
using namespace rapidjson;
|
||||
using namespace std;
|
||||
using namespace simdjson;
|
||||
|
||||
std::string rapidstringmeInsitu(char *json) {
|
||||
std::string rapid_stringme_insitu(char *json) {
|
||||
Document d;
|
||||
d.ParseInsitu(json);
|
||||
if (d.HasParseError()) {
|
||||
@@ -31,7 +32,7 @@ std::string rapidstringmeInsitu(char *json) {
|
||||
return buffer.GetString();
|
||||
}
|
||||
|
||||
std::string rapidstringme(char *json) {
|
||||
std::string rapid_stringme(char *json) {
|
||||
Document d;
|
||||
d.Parse(json);
|
||||
if (d.HasParseError()) {
|
||||
@@ -44,116 +45,152 @@ std::string rapidstringme(char *json) {
|
||||
return buffer.GetString();
|
||||
}
|
||||
|
||||
std::string simdjson_stringme(simdjson::padded_string & json) {
|
||||
std::stringstream ss;
|
||||
dom::parser parser;
|
||||
dom::element doc;
|
||||
auto error = parser.parse(json).get(doc);
|
||||
if (error) { std::cerr << error << std::endl; abort(); }
|
||||
ss << simdjson::minify(doc);
|
||||
return ss.str();
|
||||
}
|
||||
|
||||
|
||||
int main(int argc, char *argv[]) {
|
||||
int c;
|
||||
bool verbose = false;
|
||||
bool justdata = false;
|
||||
bool just_data = false;
|
||||
|
||||
while ((c = getopt (argc, argv, "vt")) != -1)
|
||||
switch (c)
|
||||
{
|
||||
case 't':
|
||||
justdata = true;
|
||||
break;
|
||||
case 'v':
|
||||
verbose = true;
|
||||
break;
|
||||
default:
|
||||
abort ();
|
||||
}
|
||||
while ((c = getopt(argc, argv, "vt")) != -1)
|
||||
switch (c) {
|
||||
case 't':
|
||||
just_data = true;
|
||||
break;
|
||||
case 'v':
|
||||
verbose = true;
|
||||
break;
|
||||
default:
|
||||
abort();
|
||||
}
|
||||
if (optind >= argc) {
|
||||
cerr << "Usage: " << argv[0] << " <jsonfile>" << endl;
|
||||
std::cerr << "Usage: " << argv[0] << " <jsonfile>" << std::endl;
|
||||
exit(1);
|
||||
}
|
||||
const char * filename = argv[optind];
|
||||
std::string_view p;
|
||||
try {
|
||||
p = get_corpus(filename);
|
||||
} catch (const std::exception& e) { // caught by reference to base
|
||||
std::cout << "Could not load the file " << filename << std::endl;
|
||||
const char *filename = argv[optind];
|
||||
simdjson::padded_string p;
|
||||
auto error = simdjson::padded_string::load(filename).get(p);
|
||||
if (error) {
|
||||
std::cerr << "Could not load the file " << filename << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
// Gigabyte: https://en.wikipedia.org/wiki/Gigabyte
|
||||
if (verbose) {
|
||||
std::cout << "Input has ";
|
||||
if (p.size() > 1024 * 1024)
|
||||
std::cout << p.size() / (1024 * 1024) << " MB ";
|
||||
else if (p.size() > 1024)
|
||||
std::cout << p.size() / 1024 << " KB ";
|
||||
if (p.size() > 1000 * 1000)
|
||||
std::cout << p.size() / (1000 * 1000) << " MB ";
|
||||
else if (p.size() > 1000)
|
||||
std::cout << p.size() / 1000 << " KB ";
|
||||
else
|
||||
std::cout << p.size() << " B ";
|
||||
std::cout << std::endl;
|
||||
}
|
||||
char *buffer = allocate_padded_buffer(p.size() + 1);
|
||||
char *buffer = simdjson::internal::allocate_padded_buffer(p.size() + 1);
|
||||
if(buffer == nullptr) {
|
||||
std::cerr << "Out of memory!" << std::endl;
|
||||
abort();
|
||||
}
|
||||
memcpy(buffer, p.data(), p.size());
|
||||
buffer[p.size()] = '\0';
|
||||
|
||||
int repeat = 50;
|
||||
int volume = p.size();
|
||||
if(justdata) {
|
||||
printf("name cycles_per_byte cycles_per_byte_err gb_per_s gb_per_s_err \n");
|
||||
size_t volume = p.size();
|
||||
if (just_data) {
|
||||
printf(
|
||||
"name cycles_per_byte cycles_per_byte_err gb_per_s gb_per_s_err \n");
|
||||
}
|
||||
size_t strlength = rapidstringme((char *)p.data()).size();
|
||||
size_t strlength = rapid_stringme((char *)p.data()).size();
|
||||
if (verbose)
|
||||
std::cout << "input length is " << p.size() << " stringified length is "
|
||||
<< strlength << std::endl;
|
||||
BEST_TIME_NOCHECK("despacing with RapidJSON", rapidstringme((char *)p.data()), , repeat, volume, !justdata);
|
||||
BEST_TIME_NOCHECK("despacing with RapidJSON Insitu", rapidstringmeInsitu((char *)buffer),
|
||||
memcpy(buffer, p.data(), p.size()), repeat, volume, !justdata);
|
||||
BEST_TIME_NOCHECK("despacing with RapidJSON",
|
||||
rapid_stringme((char *)p.data()), , repeat, volume,
|
||||
!just_data);
|
||||
BEST_TIME_NOCHECK(
|
||||
"despacing with RapidJSON Insitu", rapid_stringme_insitu((char *)buffer),
|
||||
memcpy(buffer, p.data(), p.size()), repeat, volume, !just_data);
|
||||
|
||||
BEST_TIME_NOCHECK(
|
||||
"despacing with std::minify", simdjson_stringme(p),, repeat, volume, !just_data);
|
||||
|
||||
|
||||
memcpy(buffer, p.data(), p.size());
|
||||
|
||||
size_t outlength =
|
||||
jsonminify((const uint8_t *)buffer, p.size(), (uint8_t *)buffer);
|
||||
if (verbose)
|
||||
std::cout << "jsonminify length is " << outlength << std::endl;
|
||||
|
||||
size_t outlength;
|
||||
uint8_t *cbuffer = (uint8_t *)buffer;
|
||||
BEST_TIME("jsonminify", jsonminify(cbuffer, p.size(), cbuffer), outlength,
|
||||
memcpy(buffer, p.data(), p.size()), repeat, volume, !justdata);
|
||||
printf("minisize = %zu, original size = %zu (minified down to %.2f percent of original) \n", outlength, p.size(), outlength * 100.0 / p.size());
|
||||
for (auto imple : simdjson::available_implementations) {
|
||||
if(imple->supported_by_runtime_system()) {
|
||||
BEST_TIME((std::string("simdjson->minify+")+imple->name()).c_str(), (imple->minify(cbuffer, p.size(), cbuffer, outlength) == simdjson::SUCCESS ? outlength : -1),
|
||||
outlength, memcpy(buffer, p.data(), p.size()), repeat, volume,
|
||||
!just_data);
|
||||
}
|
||||
}
|
||||
|
||||
printf("minisize = %zu, original size = %zu (minified down to %.2f percent "
|
||||
"of original) \n",
|
||||
outlength, p.size(), static_cast<double>(outlength) * 100.0 / static_cast<double>(p.size()));
|
||||
|
||||
/***
|
||||
* Is it worth it to minify before parsing?
|
||||
***/
|
||||
rapidjson::Document d;
|
||||
BEST_TIME("RapidJSON Insitu orig", d.ParseInsitu(buffer).HasParseError(), false,
|
||||
memcpy(buffer, p.data(), p.size()), repeat, volume, !justdata);
|
||||
BEST_TIME("RapidJSON Insitu orig", d.ParseInsitu(buffer).HasParseError(),
|
||||
false, memcpy(buffer, p.data(), p.size()), repeat, volume,
|
||||
!just_data);
|
||||
|
||||
char *minibuffer = allocate_padded_buffer(p.size() + 1);
|
||||
size_t minisize = jsonminify((const uint8_t *)p.data(), p.size(), (uint8_t*) minibuffer);
|
||||
minibuffer[minisize] = '\0';
|
||||
char *mini_buffer = simdjson::internal::allocate_padded_buffer(p.size() + 1);
|
||||
if(mini_buffer == nullptr) {
|
||||
std::cerr << "Out of memory" << std::endl;
|
||||
abort();
|
||||
}
|
||||
size_t minisize;
|
||||
auto minierror = minify(p.data(), p.size(),mini_buffer, minisize);
|
||||
if (!minierror) { std::cerr << minierror << std::endl; exit(1); }
|
||||
mini_buffer[minisize] = '\0';
|
||||
|
||||
BEST_TIME("RapidJSON Insitu despaced", d.ParseInsitu(buffer).HasParseError(), false,
|
||||
memcpy(buffer, minibuffer, p.size()),
|
||||
repeat, volume, !justdata);
|
||||
BEST_TIME("RapidJSON Insitu despaced", d.ParseInsitu(buffer).HasParseError(),
|
||||
false, memcpy(buffer, mini_buffer, p.size()), repeat, volume,
|
||||
!just_data);
|
||||
|
||||
size_t astbuffersize = p.size() * 2;
|
||||
size_t * ast_buffer = (size_t *) malloc(astbuffersize * sizeof(size_t));
|
||||
size_t ast_buffer_size = p.size() * 2;
|
||||
size_t *ast_buffer = (size_t *)malloc(ast_buffer_size * sizeof(size_t));
|
||||
|
||||
BEST_TIME("sajson orig", sajson::parse(sajson::bounded_allocation(ast_buffer, astbuffersize), sajson::mutable_string_view(p.size(), buffer)).is_valid(), true, memcpy(buffer, p.data(), p.size()), repeat, volume, !justdata);
|
||||
BEST_TIME(
|
||||
"sajson orig",
|
||||
sajson::parse(sajson::bounded_allocation(ast_buffer, ast_buffer_size),
|
||||
sajson::mutable_string_view(p.size(), buffer))
|
||||
.is_valid(),
|
||||
true, memcpy(buffer, p.data(), p.size()), repeat, volume, !just_data);
|
||||
|
||||
BEST_TIME(
|
||||
"sajson despaced",
|
||||
sajson::parse(sajson::bounded_allocation(ast_buffer, ast_buffer_size),
|
||||
sajson::mutable_string_view(minisize, buffer))
|
||||
.is_valid(),
|
||||
true, memcpy(buffer, mini_buffer, p.size()), repeat, volume, !just_data);
|
||||
|
||||
BEST_TIME("sajson despaced", sajson::parse(sajson::bounded_allocation(ast_buffer, astbuffersize), sajson::mutable_string_view(minisize, buffer)).is_valid(), true, memcpy(buffer, minibuffer, p.size()), repeat, volume, !justdata);
|
||||
simdjson::dom::parser parser;
|
||||
bool automated_reallocation = false;
|
||||
BEST_TIME("simdjson orig",
|
||||
parser.parse((const uint8_t *)buffer, p.size(),
|
||||
automated_reallocation).error(),
|
||||
simdjson::SUCCESS, memcpy(buffer, p.data(), p.size()), repeat, volume,
|
||||
!just_data);
|
||||
BEST_TIME("simdjson despaced",
|
||||
parser.parse((const uint8_t *)buffer, minisize,
|
||||
automated_reallocation).error(),
|
||||
simdjson::SUCCESS, memcpy(buffer, mini_buffer, p.size()), repeat, volume,
|
||||
!just_data);
|
||||
|
||||
ParsedJson pj;
|
||||
bool isallocok = pj.allocateCapacity(p.size(), 1024);
|
||||
if(!isallocok) {
|
||||
fprintf(stderr, "failed to allocate memory\n");
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
BEST_TIME("simdjson orig", json_parse((const uint8_t*)buffer, p.size(), pj), true, memcpy(buffer, p.data(), p.size()), repeat, volume, !justdata);
|
||||
|
||||
ParsedJson pj2;
|
||||
bool isallocok2 = pj2.allocateCapacity(p.size(), 1024);
|
||||
if(!isallocok2) {
|
||||
fprintf(stderr, "failed to allocate memory\n");
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
|
||||
BEST_TIME("simdjson despaced", json_parse((const uint8_t*)buffer, minisize, pj2), true, memcpy(buffer, minibuffer, p.size()), repeat, volume, !justdata);
|
||||
aligned_free((void*)p.data());
|
||||
free(buffer);
|
||||
free(ast_buffer);
|
||||
free(minibuffer);
|
||||
|
||||
|
||||
free(mini_buffer);
|
||||
}
|
||||
|
||||
+187
-234
@@ -1,12 +1,11 @@
|
||||
#include "event_counter.h"
|
||||
|
||||
#include <cassert>
|
||||
#include <cctype>
|
||||
#ifndef _MSC_VER
|
||||
#include <dirent.h>
|
||||
#include <unistd.h>
|
||||
#include <x86intrin.h>
|
||||
#else
|
||||
#include <intrin.h>
|
||||
#endif
|
||||
#include <unistd.h>
|
||||
#include <cinttypes>
|
||||
|
||||
#include <cstdio>
|
||||
@@ -29,245 +28,199 @@
|
||||
#ifdef __linux__
|
||||
#include <libgen.h>
|
||||
#endif
|
||||
//#define DEBUG
|
||||
#include "simdjson/common_defs.h"
|
||||
#include "simdjson/jsonioutil.h"
|
||||
#include "simdjson/jsonparser.h"
|
||||
#include "simdjson/parsedjson.h"
|
||||
#include "simdjson/stage1_find_marks.h"
|
||||
#include "simdjson/stage2_build_tape.h"
|
||||
using namespace std;
|
||||
|
||||
int main(int argc, char *argv[]) {
|
||||
#include "simdjson.h"
|
||||
|
||||
#include <functional>
|
||||
|
||||
#include "benchmarker.h"
|
||||
|
||||
using namespace simdjson;
|
||||
using std::cerr;
|
||||
using std::cout;
|
||||
using std::endl;
|
||||
using std::string;
|
||||
using std::to_string;
|
||||
using std::vector;
|
||||
using std::ostream;
|
||||
using std::ofstream;
|
||||
using std::exception;
|
||||
|
||||
// Stash the exe_name in main() for functions to use
|
||||
char* exe_name;
|
||||
|
||||
void print_usage(ostream& out) {
|
||||
out << "Usage: " << exe_name << " [-vt] [-n #] [-s STAGE] [-a ARCH] <jsonfile> ..." << endl;
|
||||
out << endl;
|
||||
out << "Runs the parser against the given json files in a loop, measuring speed and other statistics." << endl;
|
||||
out << endl;
|
||||
out << "Options:" << endl;
|
||||
out << endl;
|
||||
out << "-n # - Number of iterations per file. Default: 200" << endl;
|
||||
out << "-i # - Number of times to iterate a single file before moving to the next. Default: 20" << endl;
|
||||
out << "-t - Tabbed data output" << endl;
|
||||
out << "-v - Verbose output." << endl;
|
||||
out << "-s stage1 - Stop after find_structural_bits." << endl;
|
||||
out << "-s all - Run all stages." << endl;
|
||||
out << "-C - Leave the buffers cold (includes page allocation and related OS tasks during parsing, speed tied to OS performance)" << endl;
|
||||
out << "-H - Make the buffers hot (reduce page allocation and related OS tasks during parsing) [default]" << endl;
|
||||
out << "-a IMPL - Use the given parser implementation. By default, detects the most advanced" << endl;
|
||||
out << " implementation supported on the host machine." << endl;
|
||||
for (auto impl : simdjson::available_implementations) {
|
||||
if(impl->supported_by_runtime_system()) {
|
||||
out << "-a " << std::left << std::setw(9) << impl->name() << " - Use the " << impl->description() << " parser implementation." << endl;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void exit_usage(string message) {
|
||||
cerr << message << endl;
|
||||
cerr << endl;
|
||||
print_usage(cerr);
|
||||
exit(EXIT_FAILURE);
|
||||
}
|
||||
|
||||
struct option_struct {
|
||||
vector<char*> files{};
|
||||
bool stage1_only = false;
|
||||
|
||||
int32_t iterations = 200;
|
||||
int32_t iteration_step = -1;
|
||||
|
||||
bool verbose = false;
|
||||
bool dump = false;
|
||||
bool jsonoutput = false;
|
||||
bool forceoneiteration = false;
|
||||
bool justdata = false;
|
||||
#ifndef _MSC_VER
|
||||
int c;
|
||||
bool tabbed_output = false;
|
||||
/**
|
||||
* Benchmarking on a cold parser instance means that the parsing may include
|
||||
* memory allocation at the OS level. This may lead to apparently odd results
|
||||
* such that higher speed under the Windows Subsystem for Linux than under the
|
||||
* regular Windows, for the same machine. It is arguably misleading to benchmark
|
||||
* how the OS allocates memory, when we really want to just benchmark simdjson.
|
||||
*/
|
||||
bool hotbuffers = true;
|
||||
|
||||
while ((c = getopt(argc, argv, "1vdt")) != -1) {
|
||||
switch (c) {
|
||||
case 't':
|
||||
justdata = true;
|
||||
break;
|
||||
case 'v':
|
||||
verbose = true;
|
||||
break;
|
||||
case 'd':
|
||||
dump = true;
|
||||
break;
|
||||
case 'j':
|
||||
jsonoutput = true;
|
||||
break;
|
||||
case '1':
|
||||
forceoneiteration = true;
|
||||
break;
|
||||
default:
|
||||
abort();
|
||||
}
|
||||
}
|
||||
#else
|
||||
int optind = 1;
|
||||
#endif
|
||||
if (optind >= argc) {
|
||||
cerr << "Usage: " << argv[0] << " <jsonfile>" << endl;
|
||||
exit(1);
|
||||
}
|
||||
const char *filename = argv[optind];
|
||||
if (optind + 1 < argc) {
|
||||
cerr << "warning: ignoring everything after " << argv[optind + 1] << endl;
|
||||
}
|
||||
if (verbose) {
|
||||
cout << "[verbose] loading " << filename << endl;
|
||||
}
|
||||
std::string_view p;
|
||||
try {
|
||||
p = get_corpus(filename);
|
||||
} catch (const std::exception &e) { // caught by reference to base
|
||||
std::cout << "Could not load the file " << filename << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
if (verbose) {
|
||||
cout << "[verbose] loaded " << filename << " (" << p.size() << " bytes)"
|
||||
<< endl;
|
||||
}
|
||||
#if defined(DEBUG)
|
||||
const uint32_t iterations = 1;
|
||||
#else
|
||||
const uint32_t iterations =
|
||||
forceoneiteration ? 1 : (p.size() < 1 * 1000 * 1000 ? 1000 : 10);
|
||||
#endif
|
||||
vector<double> res;
|
||||
res.resize(iterations);
|
||||
option_struct(int argc, char **argv) {
|
||||
int c;
|
||||
|
||||
#if !defined(__linux__)
|
||||
#define SQUASH_COUNTERS
|
||||
if (justdata) {
|
||||
printf("justdata (-t) flag only works under linux.\n");
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifndef SQUASH_COUNTERS
|
||||
vector<int> evts;
|
||||
evts.push_back(PERF_COUNT_HW_CPU_CYCLES);
|
||||
evts.push_back(PERF_COUNT_HW_INSTRUCTIONS);
|
||||
evts.push_back(PERF_COUNT_HW_BRANCH_MISSES);
|
||||
evts.push_back(PERF_COUNT_HW_CACHE_REFERENCES);
|
||||
evts.push_back(PERF_COUNT_HW_CACHE_MISSES);
|
||||
LinuxEvents<PERF_TYPE_HARDWARE> unified(evts);
|
||||
vector<unsigned long long> results;
|
||||
results.resize(evts.size());
|
||||
unsigned long cy0 = 0, cy1 = 0, cy2 = 0;
|
||||
unsigned long cl0 = 0, cl1 = 0, cl2 = 0;
|
||||
unsigned long mis0 = 0, mis1 = 0, mis2 = 0;
|
||||
unsigned long cref0 = 0, cref1 = 0, cref2 = 0;
|
||||
unsigned long cmis0 = 0, cmis1 = 0, cmis2 = 0;
|
||||
#endif
|
||||
bool isok = true;
|
||||
|
||||
for (uint32_t i = 0; i < iterations; i++) {
|
||||
if (verbose) {
|
||||
cout << "[verbose] iteration # " << i << endl;
|
||||
}
|
||||
#ifndef SQUASH_COUNTERS
|
||||
unified.start();
|
||||
#endif
|
||||
ParsedJson pj;
|
||||
bool allocok = pj.allocateCapacity(p.size());
|
||||
if (!allocok) {
|
||||
std::cerr << "failed to allocate memory" << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
#ifndef SQUASH_COUNTERS
|
||||
unified.end(results);
|
||||
cy0 += results[0];
|
||||
cl0 += results[1];
|
||||
mis0 += results[2];
|
||||
cref0 += results[3];
|
||||
cmis0 += results[4];
|
||||
#endif
|
||||
if (verbose) {
|
||||
cout << "[verbose] allocated memory for parsed JSON " << endl;
|
||||
}
|
||||
|
||||
auto start = std::chrono::steady_clock::now();
|
||||
#ifndef SQUASH_COUNTERS
|
||||
unified.start();
|
||||
#endif
|
||||
isok = find_structural_bits(p.data(), p.size(), pj);
|
||||
#ifndef SQUASH_COUNTERS
|
||||
unified.end(results);
|
||||
cy1 += results[0];
|
||||
cl1 += results[1];
|
||||
mis1 += results[2];
|
||||
cref1 += results[3];
|
||||
cmis1 += results[4];
|
||||
if (!isok) {
|
||||
cout << "Failed out during stage 1\n";
|
||||
break;
|
||||
}
|
||||
unified.start();
|
||||
#endif
|
||||
|
||||
isok = isok && unified_machine(p.data(), p.size(), pj);
|
||||
#ifndef SQUASH_COUNTERS
|
||||
unified.end(results);
|
||||
cy2 += results[0];
|
||||
cl2 += results[1];
|
||||
mis2 += results[2];
|
||||
cref2 += results[3];
|
||||
cmis2 += results[4];
|
||||
if (!isok) {
|
||||
cout << "Failed out during stage 2\n";
|
||||
break;
|
||||
}
|
||||
#endif
|
||||
|
||||
auto end = std::chrono::steady_clock::now();
|
||||
std::chrono::duration<double> secs = end - start;
|
||||
res[i] = secs.count();
|
||||
}
|
||||
ParsedJson pj = build_parsed_json(p); // do the parsing again to get the stats
|
||||
if (!pj.isValid()) {
|
||||
std::cerr << "Could not parse. " << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
#ifndef SQUASH_COUNTERS
|
||||
unsigned long total = cy0 + cy1 + cy2;
|
||||
if (justdata) {
|
||||
float cpb0 = (double)cy0 / (iterations * p.size());
|
||||
float cpb1 = (double)cy1 / (iterations * p.size());
|
||||
float cpb2 = (double)cy2 / (iterations * p.size());
|
||||
float cpbtotal = (double)total / (iterations * p.size());
|
||||
char *newfile = (char *)malloc(strlen(filename) + 1);
|
||||
if (newfile == NULL)
|
||||
return EXIT_FAILURE;
|
||||
::strcpy(newfile, filename);
|
||||
char *snewfile = ::basename(newfile);
|
||||
size_t nl = strlen(snewfile);
|
||||
for (size_t j = nl - 1; j > 0; j--) {
|
||||
if (snewfile[j] == '.') {
|
||||
snewfile[j] = '\0';
|
||||
while ((c = getopt(argc, argv, "vtn:i:a:s:HC")) != -1) {
|
||||
switch (c) {
|
||||
case 'n':
|
||||
iterations = atoi(optarg);
|
||||
break;
|
||||
case 'i':
|
||||
iteration_step = atoi(optarg);
|
||||
break;
|
||||
case 't':
|
||||
tabbed_output = true;
|
||||
break;
|
||||
case 'v':
|
||||
verbose = true;
|
||||
break;
|
||||
case 'a': {
|
||||
const implementation *impl = simdjson::available_implementations[optarg];
|
||||
if ((!impl) || (!impl->supported_by_runtime_system())) {
|
||||
std::string exit_message = string("Unsupported option value -a ") + optarg + ": expected -a with one of ";
|
||||
for (auto imple : simdjson::available_implementations) {
|
||||
if(imple->supported_by_runtime_system()) {
|
||||
exit_message += imple->name();
|
||||
exit_message += " ";
|
||||
}
|
||||
}
|
||||
exit_usage(exit_message);
|
||||
}
|
||||
simdjson::active_implementation = impl;
|
||||
break;
|
||||
}
|
||||
case 'C':
|
||||
hotbuffers = false;
|
||||
break;
|
||||
case 'H':
|
||||
hotbuffers = true;
|
||||
break;
|
||||
case 's':
|
||||
if (!strcmp(optarg, "stage1")) {
|
||||
stage1_only = true;
|
||||
} else if (!strcmp(optarg, "all")) {
|
||||
stage1_only = false;
|
||||
} else {
|
||||
exit_usage(string("Unsupported option value -s ") + optarg + ": expected -s stage1 or all");
|
||||
}
|
||||
break;
|
||||
default:
|
||||
// reaching here means an argument was given to getopt() which did not have a case label
|
||||
exit_usage("Unexpected argument - missing case for option "+
|
||||
std::string(1,static_cast<char>(c))+
|
||||
" (programming error)");
|
||||
}
|
||||
}
|
||||
|
||||
if (iteration_step == -1) {
|
||||
iteration_step = iterations / 50;
|
||||
if (iteration_step < 200) { iteration_step = 200; }
|
||||
if (iteration_step > iterations) { iteration_step = iterations; }
|
||||
}
|
||||
|
||||
// All remaining arguments are considered to be files
|
||||
for (int i=optind; i<argc; i++) {
|
||||
files.push_back(argv[i]);
|
||||
}
|
||||
if (files.empty()) {
|
||||
exit_usage("No files specified");
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
int main(int argc, char *argv[]) {
|
||||
// Read options
|
||||
exe_name = argv[0];
|
||||
option_struct options(argc, argv);
|
||||
if (options.verbose) {
|
||||
verbose_stream = &cout;
|
||||
verbose() << "Implementation: " << simdjson::active_implementation->name() << endl;
|
||||
}
|
||||
|
||||
// Start collecting events. We put this early so if it prints an error message, it's the
|
||||
// first thing printed.
|
||||
event_collector collector;
|
||||
|
||||
// Print preamble
|
||||
if (!options.tabbed_output) {
|
||||
printf("number of iterations %u \n", options.iterations);
|
||||
}
|
||||
|
||||
// Set up benchmarkers by reading all files
|
||||
vector<benchmarker*> benchmarkers;
|
||||
for (size_t i=0; i<options.files.size(); i++) {
|
||||
benchmarkers.push_back(new benchmarker(options.files[i], collector));
|
||||
}
|
||||
|
||||
// Run the benchmarks
|
||||
progress_bar progress(options.iterations, 50);
|
||||
// Put the if (options.stage1_only) *outside* the loop so that run_iterations will be optimized
|
||||
if (options.stage1_only) {
|
||||
for (int iteration = 0; iteration < options.iterations; iteration += options.iteration_step) {
|
||||
if (!options.verbose) { progress.print(iteration); }
|
||||
// Benchmark each file once per iteration
|
||||
for (size_t f=0; f<options.files.size(); f++) {
|
||||
verbose() << "[verbose] " << benchmarkers[f]->filename << " iterations #" << iteration << "-" << (iteration+options.iteration_step-1) << endl;
|
||||
benchmarkers[f]->run_iterations(options.iteration_step, true, options.hotbuffers);
|
||||
}
|
||||
}
|
||||
printf("\"%s\"\t%f\t%f\t%f\t%f\n", snewfile, cpb0, cpb1, cpb2,
|
||||
cpbtotal);
|
||||
free(newfile);
|
||||
} else {
|
||||
printf("number of bytes %ld number of structural chars %u ratio %.3f\n",
|
||||
p.size(), pj.n_structural_indexes,
|
||||
(double)pj.n_structural_indexes / p.size());
|
||||
printf("mem alloc instructions: %10lu cycles: %10lu (%.2f %%) ins/cycles: "
|
||||
"%.2f mis. branches: %10lu (cycles/mis.branch %.2f) cache accesses: "
|
||||
"%10lu (failure %10lu)\n",
|
||||
cl0 / iterations, cy0 / iterations, 100. * cy0 / total,
|
||||
(double)cl0 / cy0, mis0 / iterations, (double)cy0 / mis0,
|
||||
cref1 / iterations, cmis0 / iterations);
|
||||
printf(" mem alloc runs at %.2f cycles per input byte.\n",
|
||||
(double)cy0 / (iterations * p.size()));
|
||||
printf("stage 1 instructions: %10lu cycles: %10lu (%.2f %%) ins/cycles: "
|
||||
"%.2f mis. branches: %10lu (cycles/mis.branch %.2f) cache accesses: "
|
||||
"%10lu (failure %10lu)\n",
|
||||
cl1 / iterations, cy1 / iterations, 100. * cy1 / total,
|
||||
(double)cl1 / cy1, mis1 / iterations, (double)cy1 / mis1,
|
||||
cref1 / iterations, cmis1 / iterations);
|
||||
printf(" stage 1 runs at %.2f cycles per input byte.\n",
|
||||
(double)cy1 / (iterations * p.size()));
|
||||
for (int iteration = 0; iteration < options.iterations; iteration += options.iteration_step) {
|
||||
if (!options.verbose) { progress.print(iteration); }
|
||||
// Benchmark each file once per iteration
|
||||
for (size_t f=0; f<options.files.size(); f++) {
|
||||
verbose() << "[verbose] " << benchmarkers[f]->filename << " iterations #" << iteration << "-" << (iteration+options.iteration_step-1) << endl;
|
||||
benchmarkers[f]->run_iterations(options.iteration_step, false, options.hotbuffers);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (!options.verbose) { progress.erase(); }
|
||||
|
||||
printf("stage 2 instructions: %10lu cycles: %10lu (%.2f %%) ins/cycles: "
|
||||
"%.2f mis. branches: %10lu (cycles/mis.branch %.2f) cache "
|
||||
"accesses: %10lu (failure %10lu)\n",
|
||||
cl2 / iterations, cy2 / iterations, 100. * cy2 / total,
|
||||
(double)cl2 / cy2, mis2 / iterations, (double)cy2 / mis2,
|
||||
cref2 / iterations, cmis2 / iterations);
|
||||
printf(" stage 2 runs at %.2f cycles per input byte and ",
|
||||
(double)cy2 / (iterations * p.size()));
|
||||
printf("%.2f cycles per structural character.\n",
|
||||
(double)cy2 / (iterations * pj.n_structural_indexes));
|
||||
for (size_t i=0; i<options.files.size(); i++) {
|
||||
benchmarkers[i]->print(options.tabbed_output);
|
||||
delete benchmarkers[i];
|
||||
}
|
||||
|
||||
printf(" all stages: %.2f cycles per input byte.\n",
|
||||
(double)total / (iterations * p.size()));
|
||||
}
|
||||
#endif
|
||||
double min_result = *min_element(res.begin(), res.end());
|
||||
if (!justdata) {
|
||||
cout << "Min: " << min_result << " bytes read: " << p.size()
|
||||
<< " Gigabytes/second: " << (p.size()) / (min_result * 1000000000.0)
|
||||
<< "\n";
|
||||
}
|
||||
if (jsonoutput) {
|
||||
isok = isok && pj.printjson(std::cout);
|
||||
}
|
||||
if (dump) {
|
||||
isok = isok && pj.dump_raw_tape(std::cout);
|
||||
}
|
||||
aligned_free((void *)p.data());
|
||||
if (!isok) {
|
||||
fprintf(stderr, " Parsing failed. \n ");
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
return EXIT_SUCCESS;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,191 @@
|
||||
#include <algorithm>
|
||||
#include <chrono>
|
||||
#include <iostream>
|
||||
#include <map>
|
||||
#include <vector>
|
||||
|
||||
#include "simdjson.h"
|
||||
|
||||
#define NB_ITERATION 20
|
||||
#define MIN_BATCH_SIZE 10000
|
||||
#define MAX_BATCH_SIZE 10000000
|
||||
|
||||
bool test_baseline = false;
|
||||
bool test_per_batch = true;
|
||||
bool test_best_batch = false;
|
||||
|
||||
bool compare(std::pair<size_t, double> i, std::pair<size_t, double> j) {
|
||||
return i.second > j.second;
|
||||
}
|
||||
|
||||
int main(int argc, char *argv[]) {
|
||||
|
||||
if (argc <= 1) {
|
||||
std::cerr << "Usage: " << argv[0] << " <jsonfile>" << std::endl;
|
||||
exit(1);
|
||||
}
|
||||
const char *filename = argv[1];
|
||||
auto[p, err] = simdjson::padded_string::load(filename);
|
||||
if (err) {
|
||||
std::cerr << "Could not load the file " << filename << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
if (test_baseline) {
|
||||
std::wclog << "Baseline: Getline + normal parse... " << std::endl;
|
||||
std::cout << "Gigabytes/second\t"
|
||||
<< "Nb of documents parsed" << std::endl;
|
||||
for (auto i = 0; i < 3; i++) {
|
||||
// Actual test
|
||||
simdjson::dom::parser parser;
|
||||
simdjson::error_code alloc_error = parser.allocate(p.size());
|
||||
if (alloc_error) {
|
||||
std::cerr << alloc_error << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
std::istringstream ss(std::string(p.data(), p.size()));
|
||||
|
||||
auto start = std::chrono::steady_clock::now();
|
||||
int count = 0;
|
||||
std::string line;
|
||||
int parse_res = simdjson::SUCCESS;
|
||||
while (getline(ss, line)) {
|
||||
// TODO we're likely triggering simdjson's padding reallocation here. Is
|
||||
// that intentional?
|
||||
parser.parse(line);
|
||||
count++;
|
||||
}
|
||||
|
||||
auto end = std::chrono::steady_clock::now();
|
||||
|
||||
std::chrono::duration<double> secs = end - start;
|
||||
double speedinGBs = static_cast<double>(p.size()) /
|
||||
(static_cast<double>(secs.count()) * 1000000000.0);
|
||||
std::cout << speedinGBs << "\t\t\t\t" << count << std::endl;
|
||||
|
||||
if (parse_res != simdjson::SUCCESS) {
|
||||
std::cerr << "Parsing failed" << std::endl;
|
||||
exit(1);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
std::map<size_t, double> batch_size_res;
|
||||
if (test_per_batch) {
|
||||
std::wclog << "parse_many: Speed per batch_size... from " << MIN_BATCH_SIZE
|
||||
<< " bytes to " << MAX_BATCH_SIZE << " bytes..." << std::endl;
|
||||
std::cout << "Batch Size\t"
|
||||
<< "Gigabytes/second\t"
|
||||
<< "Nb of documents parsed" << std::endl;
|
||||
for (size_t i = MIN_BATCH_SIZE; i <= MAX_BATCH_SIZE;
|
||||
i += (MAX_BATCH_SIZE - MIN_BATCH_SIZE) / 100) {
|
||||
batch_size_res.insert(std::pair<size_t, double>(i, 0));
|
||||
int count;
|
||||
for (size_t j = 0; j < 5; j++) {
|
||||
// Actual test
|
||||
simdjson::dom::parser parser;
|
||||
simdjson::error_code error;
|
||||
|
||||
auto start = std::chrono::steady_clock::now();
|
||||
count = 0;
|
||||
simdjson::dom::document_stream docs;
|
||||
if ((error = parser.parse_many(p, i).get(docs))) {
|
||||
std::wcerr << "Parsing failed with: " << error << std::endl;
|
||||
exit(1);
|
||||
}
|
||||
for (auto result : docs) {
|
||||
error = result.error();
|
||||
if (error) {
|
||||
std::wcerr << "Parsing failed with: " << error << std::endl;
|
||||
exit(1);
|
||||
}
|
||||
count++;
|
||||
}
|
||||
auto end = std::chrono::steady_clock::now();
|
||||
|
||||
std::chrono::duration<double> secs = end - start;
|
||||
double speedinGBs = static_cast<double>(p.size()) /
|
||||
(static_cast<double>(secs.count()) * 1000000000.0);
|
||||
if (speedinGBs > batch_size_res.at(i))
|
||||
batch_size_res[i] = speedinGBs;
|
||||
}
|
||||
std::cout << i << "\t\t" << std::fixed << std::setprecision(3)
|
||||
<< batch_size_res.at(i) << "\t\t\t\t" << count << std::endl;
|
||||
}
|
||||
}
|
||||
size_t optimal_batch_size{};
|
||||
double best_speed{};
|
||||
if (test_per_batch) {
|
||||
std::pair<size_t, double> best_results;
|
||||
best_results =
|
||||
(*min_element(batch_size_res.begin(), batch_size_res.end(), compare));
|
||||
optimal_batch_size = best_results.first;
|
||||
best_speed = best_results.second;
|
||||
} else {
|
||||
optimal_batch_size = MIN_BATCH_SIZE;
|
||||
}
|
||||
std::wclog << "Seemingly optimal batch_size: " << optimal_batch_size << "..."
|
||||
<< std::endl;
|
||||
std::wclog << "Best speed: " << best_speed << "..." << std::endl;
|
||||
|
||||
if (test_best_batch) {
|
||||
std::wclog << "Starting speed test... Best of " << NB_ITERATION
|
||||
<< " iterations..." << std::endl;
|
||||
std::vector<double> res;
|
||||
for (int i = 0; i < NB_ITERATION; i++) {
|
||||
|
||||
// Actual test
|
||||
simdjson::dom::parser parser;
|
||||
simdjson::error_code error;
|
||||
|
||||
auto start = std::chrono::steady_clock::now();
|
||||
// This includes allocation of the parser
|
||||
simdjson::dom::document_stream docs;
|
||||
if ((error = parser.parse_many(p, optimal_batch_size).get(docs))) {
|
||||
std::wcerr << "Parsing failed with: " << error << std::endl;
|
||||
exit(1);
|
||||
}
|
||||
for (auto result : docs) {
|
||||
error = result.error();
|
||||
if (error) {
|
||||
std::wcerr << "Parsing failed with: " << error << std::endl;
|
||||
exit(1);
|
||||
}
|
||||
}
|
||||
auto end = std::chrono::steady_clock::now();
|
||||
|
||||
std::chrono::duration<double> secs = end - start;
|
||||
res.push_back(secs.count());
|
||||
}
|
||||
|
||||
double min_result = *min_element(res.begin(), res.end());
|
||||
double speedinGBs =
|
||||
static_cast<double>(p.size()) / (min_result * 1000000000.0);
|
||||
|
||||
std::cout << "Min: " << min_result << " bytes read: " << p.size()
|
||||
<< " Gigabytes/second: " << speedinGBs << std::endl;
|
||||
}
|
||||
#ifdef SIMDJSON_THREADS_ENABLED
|
||||
// Multithreading probably does not help matters for small files (less than 10
|
||||
// MB).
|
||||
if (p.size() < 10000000) {
|
||||
std::cout << std::endl;
|
||||
|
||||
std::cout << "Warning: your file is small and the performance results are "
|
||||
"probably meaningless"
|
||||
<< std::endl;
|
||||
std::cout << "as far as multithreaded performance goes." << std::endl;
|
||||
|
||||
std::cout << std::endl;
|
||||
|
||||
std::cout
|
||||
<< "Try to concatenate the file with itself to generate a large one."
|
||||
<< std::endl;
|
||||
std::cout << "In bash: " << std::endl;
|
||||
std::cout << "for i in {1..1000}; do cat '" << filename
|
||||
<< "' >> bar.ndjson; done" << std::endl;
|
||||
std::cout << argv[0] << " bar.ndjson" << std::endl;
|
||||
}
|
||||
#endif
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -1,7 +1,10 @@
|
||||
#include "simdjson/jsonparser.h"
|
||||
#include "simdjson.h"
|
||||
#include <unistd.h>
|
||||
|
||||
#include "benchmark.h"
|
||||
|
||||
SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
|
||||
|
||||
// #define RAPIDJSON_SSE2 // bad for performance
|
||||
// #define RAPIDJSON_SSE42 // bad for performance
|
||||
#include "rapidjson/document.h"
|
||||
@@ -11,9 +14,10 @@
|
||||
|
||||
#include "sajson.h"
|
||||
|
||||
using namespace rapidjson;
|
||||
using namespace std;
|
||||
SIMDJSON_POP_DISABLE_WARNINGS
|
||||
|
||||
using namespace rapidjson;
|
||||
using namespace simdjson;
|
||||
struct stat_s {
|
||||
size_t number_count;
|
||||
size_t object_count;
|
||||
@@ -44,63 +48,196 @@ void print_stat(const stat_t &s) {
|
||||
s.true_count, s.false_count);
|
||||
}
|
||||
|
||||
__attribute__ ((noinline))
|
||||
stat_t simdjson_computestats(const std::string_view &p) {
|
||||
stat_t answer;
|
||||
ParsedJson pj = build_parsed_json(p);
|
||||
answer.valid = pj.isValid();
|
||||
if (!answer.valid) {
|
||||
return answer;
|
||||
}
|
||||
answer.number_count = 0;
|
||||
answer.object_count = 0;
|
||||
answer.array_count = 0;
|
||||
answer.null_count = 0;
|
||||
answer.true_count = 0;
|
||||
answer.false_count = 0;
|
||||
size_t tapeidx = 0;
|
||||
uint64_t tape_val = pj.tape[tapeidx++];
|
||||
uint8_t type = (tape_val >> 56);
|
||||
size_t howmany = 0;
|
||||
assert(type == 'r');
|
||||
howmany = tape_val & JSONVALUEMASK;
|
||||
for (; tapeidx < howmany; tapeidx++) {
|
||||
tape_val = pj.tape[tapeidx];
|
||||
// uint64_t payload = tape_val & JSONVALUEMASK;
|
||||
type = (tape_val >> 56);
|
||||
switch (type) {
|
||||
case 'l': // we have a long int
|
||||
answer.number_count++;
|
||||
tapeidx++; // skipping the integer
|
||||
break;
|
||||
case 'd': // we have a double
|
||||
answer.number_count++;
|
||||
tapeidx++; // skipping the double
|
||||
break;
|
||||
case 'n': // we have a null
|
||||
answer.null_count++;
|
||||
break;
|
||||
case 't': // we have a true
|
||||
answer.true_count++;
|
||||
break;
|
||||
case 'f': // we have a false
|
||||
answer.false_count++;
|
||||
break;
|
||||
case '{': // we have an object
|
||||
answer.object_count++;
|
||||
break;
|
||||
case '}': // we end an object
|
||||
break;
|
||||
case '[': // we start an array
|
||||
answer.array_count++;
|
||||
break;
|
||||
case ']': // we end an array
|
||||
break;
|
||||
default:
|
||||
break; // ignore
|
||||
simdjson_really_inline void simdjson_process_atom(stat_t &s,
|
||||
simdjson::dom::element element) {
|
||||
if (element.is<double>()) {
|
||||
s.number_count++;
|
||||
} else if (element.is<bool>()) {
|
||||
simdjson::error_code error;
|
||||
bool v;
|
||||
if (not (error = element.get(v)) && v) {
|
||||
s.true_count++;
|
||||
} else {
|
||||
s.false_count++;
|
||||
}
|
||||
} else if (element.is_null()) {
|
||||
s.null_count++;
|
||||
}
|
||||
return answer;
|
||||
}
|
||||
|
||||
void simdjson_recurse(stat_t &s, simdjson::dom::element element) {
|
||||
error_code error;
|
||||
if (element.is<simdjson::dom::array>()) {
|
||||
s.array_count++;
|
||||
dom::array array;
|
||||
if ((error = element.get(array))) {
|
||||
std::cerr << error << std::endl;
|
||||
abort();
|
||||
}
|
||||
for (auto child : array) {
|
||||
if (child.is<simdjson::dom::array>() ||
|
||||
child.is<simdjson::dom::object>()) {
|
||||
simdjson_recurse(s, child);
|
||||
} else {
|
||||
simdjson_process_atom(s, child);
|
||||
}
|
||||
}
|
||||
} else if (element.is<simdjson::dom::object>()) {
|
||||
s.object_count++;
|
||||
dom::object object;
|
||||
if ((error = element.get(object))) {
|
||||
std::cerr << error << std::endl;
|
||||
abort();
|
||||
}
|
||||
for (auto field : object) {
|
||||
if (field.value.is<simdjson::dom::array>() ||
|
||||
field.value.is<simdjson::dom::object>()) {
|
||||
simdjson_recurse(s, field.value);
|
||||
} else {
|
||||
simdjson_process_atom(s, field.value);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
simdjson_process_atom(s, element);
|
||||
}
|
||||
}
|
||||
|
||||
simdjson_never_inline stat_t simdjson_compute_stats(const simdjson::padded_string &p) {
|
||||
stat_t s{};
|
||||
simdjson::dom::parser parser;
|
||||
simdjson::dom::element doc;
|
||||
auto error = parser.parse(p).get(doc);
|
||||
if (error) {
|
||||
s.valid = false;
|
||||
return s;
|
||||
}
|
||||
s.valid = true;
|
||||
simdjson_recurse(s, doc);
|
||||
return s;
|
||||
}
|
||||
|
||||
///
|
||||
struct Stat {
|
||||
size_t objectCount;
|
||||
size_t arrayCount;
|
||||
size_t numberCount;
|
||||
size_t stringCount;
|
||||
size_t trueCount;
|
||||
size_t falseCount;
|
||||
size_t nullCount;
|
||||
|
||||
size_t memberCount; // Number of members in all objects
|
||||
size_t elementCount; // Number of elements in all arrays
|
||||
size_t stringLength; // Number of code units in all strings
|
||||
};
|
||||
|
||||
static error_code GenStatPlus(Stat &stat, const dom::element &v);
|
||||
static error_code GenStatPlus(Stat &stat, const simdjson_result<dom::element> &r) {
|
||||
dom::element v;
|
||||
SIMDJSON_TRY( r.get(v) );
|
||||
return GenStatPlus(stat, v);
|
||||
}
|
||||
static error_code GenStatPlus(Stat &stat, const dom::element &v) {
|
||||
switch (v.type()) {
|
||||
case dom::element_type::ARRAY: {
|
||||
dom::array a;
|
||||
SIMDJSON_TRY( v.get(a) )
|
||||
for (auto child : a) {
|
||||
GenStatPlus(stat, child);
|
||||
stat.elementCount++;
|
||||
}
|
||||
stat.arrayCount++;
|
||||
} break;
|
||||
case dom::element_type::OBJECT: {
|
||||
dom::object o;
|
||||
SIMDJSON_TRY( v.get(o) );
|
||||
for (dom::key_value_pair kv : o) {
|
||||
GenStatPlus(stat, kv.value);
|
||||
stat.stringLength += kv.key.size();
|
||||
stat.memberCount++;
|
||||
stat.stringCount++;
|
||||
}
|
||||
stat.objectCount++;
|
||||
} break;
|
||||
case dom::element_type::INT64:
|
||||
case dom::element_type::UINT64:
|
||||
case dom::element_type::DOUBLE:
|
||||
stat.numberCount++;
|
||||
break;
|
||||
case dom::element_type::STRING: {
|
||||
stat.stringCount++;
|
||||
std::string_view sv;
|
||||
SIMDJSON_TRY( v.get(sv) );
|
||||
stat.stringLength += sv.size();
|
||||
} break;
|
||||
case dom::element_type::BOOL: {
|
||||
bool b;
|
||||
SIMDJSON_TRY( v.get(b) );
|
||||
if (b) {
|
||||
stat.trueCount++;
|
||||
} else {
|
||||
stat.falseCount++;
|
||||
}
|
||||
} break;
|
||||
case dom::element_type::NULL_VALUE:
|
||||
++stat.nullCount;
|
||||
break;
|
||||
}
|
||||
return SUCCESS;
|
||||
}
|
||||
|
||||
static void RapidGenStat(Stat &stat, const rapidjson::Value &v) {
|
||||
switch (v.GetType()) {
|
||||
case kNullType:
|
||||
stat.nullCount++;
|
||||
break;
|
||||
case kFalseType:
|
||||
stat.falseCount++;
|
||||
break;
|
||||
case kTrueType:
|
||||
stat.trueCount++;
|
||||
break;
|
||||
|
||||
case kObjectType:
|
||||
for (Value::ConstMemberIterator m = v.MemberBegin(); m != v.MemberEnd();
|
||||
++m) {
|
||||
stat.stringLength += m->name.GetStringLength();
|
||||
RapidGenStat(stat, m->value);
|
||||
}
|
||||
stat.objectCount++;
|
||||
stat.memberCount += (v.MemberEnd() - v.MemberBegin());
|
||||
stat.stringCount += (v.MemberEnd() - v.MemberBegin()); // Key
|
||||
break;
|
||||
|
||||
case kArrayType:
|
||||
for (Value::ConstValueIterator i = v.Begin(); i != v.End(); ++i)
|
||||
RapidGenStat(stat, *i);
|
||||
stat.arrayCount++;
|
||||
stat.elementCount += v.Size();
|
||||
break;
|
||||
|
||||
case kStringType:
|
||||
stat.stringCount++;
|
||||
stat.stringLength += v.GetStringLength();
|
||||
break;
|
||||
|
||||
case kNumberType:
|
||||
stat.numberCount++;
|
||||
break;
|
||||
}
|
||||
}
|
||||
simdjson_never_inline Stat rapidjson_compute_stats_ref(const rapidjson::Value &doc) {
|
||||
Stat s{};
|
||||
RapidGenStat(s, doc);
|
||||
return s;
|
||||
}
|
||||
|
||||
simdjson_never_inline Stat
|
||||
simdjson_compute_stats_refplus(const simdjson::dom::element &doc) {
|
||||
Stat s{};
|
||||
auto error = GenStatPlus(s, doc);
|
||||
if (error) { std::cerr << error << std::endl; abort(); }
|
||||
return s;
|
||||
}
|
||||
|
||||
// see
|
||||
@@ -146,15 +283,18 @@ void sajson_traverse(stat_t &stats, const sajson::value &node) {
|
||||
}
|
||||
}
|
||||
|
||||
__attribute__ ((noinline))
|
||||
stat_t sasjon_computestats(const std::string_view &p) {
|
||||
stat_t answer;
|
||||
simdjson_never_inline stat_t sasjon_compute_stats(const simdjson::padded_string &p) {
|
||||
stat_t answer{};
|
||||
char *buffer = (char *)malloc(p.size());
|
||||
if (buffer == nullptr) {
|
||||
return answer;
|
||||
}
|
||||
memcpy(buffer, p.data(), p.size());
|
||||
auto d = sajson::parse(sajson::dynamic_allocation(),
|
||||
sajson::mutable_string_view(p.size(), buffer));
|
||||
answer.valid = d.is_valid();
|
||||
if (!answer.valid) {
|
||||
free(buffer);
|
||||
return answer;
|
||||
}
|
||||
answer.number_count = 0;
|
||||
@@ -204,16 +344,19 @@ void rapid_traverse(stat_t &stats, const rapidjson::Value &v) {
|
||||
}
|
||||
}
|
||||
|
||||
__attribute__ ((noinline))
|
||||
stat_t rapid_computestats(const std::string_view &p) {
|
||||
stat_t answer;
|
||||
simdjson_never_inline stat_t rapid_compute_stats(const simdjson::padded_string &p) {
|
||||
stat_t answer{};
|
||||
char *buffer = (char *)malloc(p.size() + 1);
|
||||
if (buffer == nullptr) {
|
||||
return answer;
|
||||
}
|
||||
memcpy(buffer, p.data(), p.size());
|
||||
buffer[p.size()] = '\0';
|
||||
rapidjson::Document d;
|
||||
d.ParseInsitu<kParseValidateEncodingFlag>(buffer);
|
||||
answer.valid = !d.HasParseError();
|
||||
if (!answer.valid) {
|
||||
free(buffer);
|
||||
return answer;
|
||||
}
|
||||
answer.number_count = 0;
|
||||
@@ -227,15 +370,41 @@ stat_t rapid_computestats(const std::string_view &p) {
|
||||
return answer;
|
||||
}
|
||||
|
||||
simdjson_never_inline stat_t
|
||||
rapid_accurate_compute_stats(const simdjson::padded_string &p) {
|
||||
stat_t answer{};
|
||||
char *buffer = (char *)malloc(p.size() + 1);
|
||||
if (buffer == nullptr) {
|
||||
return answer;
|
||||
}
|
||||
memcpy(buffer, p.data(), p.size());
|
||||
buffer[p.size()] = '\0';
|
||||
rapidjson::Document d;
|
||||
d.ParseInsitu<kParseValidateEncodingFlag | kParseFullPrecisionFlag>(buffer);
|
||||
answer.valid = !d.HasParseError();
|
||||
if (!answer.valid) {
|
||||
free(buffer);
|
||||
return answer;
|
||||
}
|
||||
answer.number_count = 0;
|
||||
answer.object_count = 0;
|
||||
answer.array_count = 0;
|
||||
answer.null_count = 0;
|
||||
answer.true_count = 0;
|
||||
answer.false_count = 0;
|
||||
rapid_traverse(answer, d);
|
||||
free(buffer);
|
||||
return answer;
|
||||
}
|
||||
int main(int argc, char *argv[]) {
|
||||
bool verbose = false;
|
||||
bool justdata = false;
|
||||
bool just_data = false;
|
||||
|
||||
int c;
|
||||
while ((c = getopt(argc, argv, "vt")) != -1)
|
||||
switch (c) {
|
||||
case 't':
|
||||
justdata = true;
|
||||
just_data = true;
|
||||
break;
|
||||
case 'v':
|
||||
verbose = true;
|
||||
@@ -244,45 +413,52 @@ int main(int argc, char *argv[]) {
|
||||
abort();
|
||||
}
|
||||
if (optind >= argc) {
|
||||
cerr << "Using different parsers, we compute the content statistics of "
|
||||
"JSON documents.\n";
|
||||
cerr << "Usage: " << argv[0] << " <jsonfile>\n";
|
||||
cerr << "Or " << argv[0] << " -v <jsonfile>\n";
|
||||
std::cerr
|
||||
<< "Using different parsers, we compute the content statistics of "
|
||||
"JSON documents."
|
||||
<< std::endl;
|
||||
std::cerr << "Usage: " << argv[0] << " <jsonfile>" << std::endl;
|
||||
std::cerr << "Or " << argv[0] << " -v <jsonfile>" << std::endl;
|
||||
exit(1);
|
||||
}
|
||||
const char *filename = argv[optind];
|
||||
if (optind + 1 < argc) {
|
||||
cerr << "warning: ignoring everything after " << argv[optind + 1] << endl;
|
||||
std::cerr << "warning: ignoring everything after " << argv[optind + 1]
|
||||
<< std::endl;
|
||||
}
|
||||
std::string_view p;
|
||||
try {
|
||||
p = get_corpus(filename);
|
||||
} catch (const std::exception &e) { // caught by reference to base
|
||||
std::cout << "Could not load the file " << filename << std::endl;
|
||||
simdjson::padded_string p;
|
||||
auto error = simdjson::padded_string::load(filename).get(p);
|
||||
if (error) {
|
||||
std::cerr << "Could not load the file " << filename << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
|
||||
// Gigabyte: https://en.wikipedia.org/wiki/Gigabyte
|
||||
if (verbose) {
|
||||
std::cout << "Input has ";
|
||||
if (p.size() > 1024 * 1024)
|
||||
std::cout << p.size() / (1024 * 1024) << " MB ";
|
||||
else if (p.size() > 1024)
|
||||
std::cout << p.size() / 1024 << " KB ";
|
||||
if (p.size() > 1000 * 1000)
|
||||
std::cout << p.size() / (1000 * 1000) << " MB ";
|
||||
else if (p.size() > 1000)
|
||||
std::cout << p.size() / 1000 << " KB ";
|
||||
else
|
||||
std::cout << p.size() << " B ";
|
||||
std::cout << std::endl;
|
||||
}
|
||||
stat_t s1 = simdjson_computestats(p);
|
||||
stat_t s1 = simdjson_compute_stats(p);
|
||||
if (verbose) {
|
||||
printf("simdjson: ");
|
||||
print_stat(s1);
|
||||
}
|
||||
stat_t s2 = rapid_computestats(p);
|
||||
stat_t s2 = rapid_compute_stats(p);
|
||||
if (verbose) {
|
||||
printf("rapid: ");
|
||||
print_stat(s2);
|
||||
}
|
||||
stat_t s3 = sasjon_computestats(p);
|
||||
stat_t s2a = rapid_accurate_compute_stats(p);
|
||||
if (verbose) {
|
||||
printf("rapid full: ");
|
||||
print_stat(s2a);
|
||||
}
|
||||
stat_t s3 = sasjon_compute_stats(p);
|
||||
if (verbose) {
|
||||
printf("sasjon: ");
|
||||
print_stat(s3);
|
||||
@@ -290,15 +466,37 @@ int main(int argc, char *argv[]) {
|
||||
assert(stat_equal(s1, s2));
|
||||
assert(stat_equal(s1, s3));
|
||||
int repeat = 50;
|
||||
int volume = p.size();
|
||||
if(justdata) {
|
||||
printf("name cycles_per_byte cycles_per_byte_err gb_per_s gb_per_s_err \n");
|
||||
size_t volume = p.size();
|
||||
if (just_data) {
|
||||
printf("name cycles_per_byte cycles_per_byte_err gb_per_s gb_per_s_err \n");
|
||||
}
|
||||
BEST_TIME("simdjson ", simdjson_compute_stats(p).valid, true, ,
|
||||
repeat, volume, !just_data);
|
||||
BEST_TIME("RapidJSON ", rapid_compute_stats(p).valid, true, ,
|
||||
repeat, volume, !just_data);
|
||||
BEST_TIME("RapidJSON (precise) ", rapid_accurate_compute_stats(p).valid, true,
|
||||
, repeat, volume, !just_data);
|
||||
BEST_TIME("sasjon ", sasjon_compute_stats(p).valid, true, ,
|
||||
repeat, volume, !just_data);
|
||||
if (!just_data) {
|
||||
printf("API traversal tests\n");
|
||||
printf("Based on https://github.com/miloyip/nativejson-benchmark\n");
|
||||
simdjson::dom::parser parser;
|
||||
simdjson::dom::element doc;
|
||||
auto error = parser.parse(p).get(doc);
|
||||
if (error) { std::cerr << error << std::endl; abort(); }
|
||||
size_t refval = simdjson_compute_stats_refplus(doc).objectCount;
|
||||
|
||||
BEST_TIME("simdjson ",
|
||||
simdjson_compute_stats_refplus(doc).objectCount, refval, , repeat,
|
||||
volume, !just_data);
|
||||
char *buffer = (char *)malloc(p.size() + 1);
|
||||
memcpy(buffer, p.data(), p.size());
|
||||
buffer[p.size()] = '\0';
|
||||
rapidjson::Document d;
|
||||
d.ParseInsitu<kParseValidateEncodingFlag>(buffer);
|
||||
BEST_TIME("rapid ", rapidjson_compute_stats_ref(d).objectCount,
|
||||
refval, , repeat, volume, !just_data);
|
||||
free(buffer);
|
||||
}
|
||||
BEST_TIME("simdjson ", simdjson_computestats(p).valid, true, , repeat,
|
||||
volume, !justdata);
|
||||
BEST_TIME("RapidJSON ", rapid_computestats(p).valid, true, , repeat, volume,
|
||||
!justdata);
|
||||
BEST_TIME("sasjon ", sasjon_computestats(p).valid, true, , repeat, volume,
|
||||
!justdata);
|
||||
aligned_free((void*)p.data());
|
||||
}
|
||||
|
||||
+365
-202
@@ -1,6 +1,7 @@
|
||||
#include "simdjson/jsonparser.h"
|
||||
#ifndef _MSC_VER
|
||||
#include "simdjson.h"
|
||||
|
||||
#include <unistd.h>
|
||||
#ifndef _MSC_VER
|
||||
#include "linux-perf-events.h"
|
||||
#ifdef __linux__
|
||||
#include <libgen.h>
|
||||
@@ -11,6 +12,8 @@
|
||||
|
||||
#include "benchmark.h"
|
||||
|
||||
SIMDJSON_PUSH_DISABLE_ALL_WARNINGS
|
||||
#include "yyjson.h"
|
||||
|
||||
// #define RAPIDJSON_SSE2 // bad for performance
|
||||
// #define RAPIDJSON_SSE42 // bad for performance
|
||||
@@ -21,35 +24,44 @@
|
||||
|
||||
#include "sajson.h"
|
||||
|
||||
#include <nlohmann/json.hpp>
|
||||
using json = nlohmann::json;
|
||||
|
||||
#ifdef HAS_BOOST_JSON
|
||||
#include <boost/json/parser.hpp>
|
||||
#include <boost/json/monotonic_resource.hpp>
|
||||
#endif
|
||||
|
||||
#ifdef ALLPARSER
|
||||
|
||||
#include "fastjson/core.h"
|
||||
#include "fastjson/dom.h"
|
||||
#include "fastjson/fastjson.h"
|
||||
|
||||
#include "fastjson.cpp"
|
||||
#include "fastjson_dom.cpp"
|
||||
#include "gason.cpp"
|
||||
#include "gason.h"
|
||||
|
||||
#include "json11.hpp"
|
||||
|
||||
#include "json11.cpp"
|
||||
extern "C" {
|
||||
#include "ujdecode.h"
|
||||
#include "ultrajsondec.c"
|
||||
#include "cJSON.h"
|
||||
#include "cJSON.c"
|
||||
|
||||
#include "jsmn.h"
|
||||
#include "jsmn.c"
|
||||
|
||||
#include "ujdecode.h"
|
||||
extern "C" {
|
||||
#include "ultrajson.h"
|
||||
}
|
||||
|
||||
#include "json/json.h"
|
||||
#include "jsoncpp.cpp"
|
||||
|
||||
#endif
|
||||
#endif
|
||||
|
||||
SIMDJSON_POP_DISABLE_WARNINGS
|
||||
|
||||
using namespace rapidjson;
|
||||
using namespace std;
|
||||
|
||||
|
||||
#ifdef ALLPARSER
|
||||
// fastjson has a tricky interface
|
||||
void on_json_error(void *, const fastjson::ErrorContext &ec) {
|
||||
void on_json_error(void *, simdjson_unused const fastjson::ErrorContext &ec) {
|
||||
// std::cerr<<"ERROR: "<<ec.mesg<<std::endl;
|
||||
}
|
||||
bool fastjson_parse(const char *input) {
|
||||
@@ -61,14 +73,335 @@ bool fastjson_parse(const char *input) {
|
||||
// end of fastjson stuff
|
||||
#endif
|
||||
|
||||
simdjson_never_inline size_t sum_line_lengths(std::stringstream &is) {
|
||||
std::string line;
|
||||
size_t sumofalllinelengths{0};
|
||||
while (std::getline(is, line)) {
|
||||
sumofalllinelengths += line.size();
|
||||
}
|
||||
return sumofalllinelengths;
|
||||
}
|
||||
|
||||
inline void reset_stream(std::stringstream &is) {
|
||||
is.clear();
|
||||
is.seekg(0, std::ios::beg);
|
||||
}
|
||||
|
||||
bool bench(const char *filename, bool verbose, bool just_data,
|
||||
double repeat_multiplier) {
|
||||
simdjson::padded_string p;
|
||||
auto error = simdjson::padded_string::load(filename).get(p);
|
||||
if (error) {
|
||||
std::cerr << "Could not load the file " << filename << ": " << error
|
||||
<< std::endl;
|
||||
return false;
|
||||
}
|
||||
|
||||
int repeat = static_cast<int>((50000000 * repeat_multiplier) /
|
||||
static_cast<double>(p.size()));
|
||||
if (repeat < 10) {
|
||||
repeat = 10;
|
||||
}
|
||||
// Gigabyte: https://en.wikipedia.org/wiki/Gigabyte
|
||||
if (verbose) {
|
||||
std::cout << "Input " << filename << " has ";
|
||||
if (p.size() > 1000 * 1000)
|
||||
std::cout << p.size() / (1000 * 1000) << " MB";
|
||||
else if (p.size() > 1000)
|
||||
std::cout << p.size() / 1000 << " KB";
|
||||
else
|
||||
std::cout << p.size() << " B";
|
||||
std::cout << ": will run " << repeat << " iterations." << std::endl;
|
||||
}
|
||||
size_t volume = p.size();
|
||||
if (just_data) {
|
||||
std::printf("%-42s %20s %20s %20s %20s \n", "name", "cycles_per_byte",
|
||||
"cycles_per_byte_err", "gb_per_s", "gb_per_s_err");
|
||||
}
|
||||
if (!just_data) {
|
||||
const std::string inputcopy(p.data(), p.data() + p.size());
|
||||
std::stringstream is;
|
||||
is.str(inputcopy);
|
||||
const size_t lc = sum_line_lengths(is);
|
||||
BEST_TIME("getline ", sum_line_lengths(is), lc, reset_stream(is), repeat,
|
||||
volume, !just_data);
|
||||
}
|
||||
|
||||
if (!just_data) {
|
||||
auto parse_dynamic = [](auto &str) {
|
||||
simdjson::dom::parser parser;
|
||||
return parser.parse(str).error();
|
||||
};
|
||||
BEST_TIME("simdjson (dynamic mem) ", parse_dynamic(p), simdjson::SUCCESS, ,
|
||||
repeat, volume, !just_data);
|
||||
}
|
||||
// (static alloc)
|
||||
simdjson::dom::parser parser;
|
||||
BEST_TIME("simdjson ", parser.parse(p).error(), simdjson::SUCCESS, , repeat,
|
||||
volume, !just_data);
|
||||
|
||||
rapidjson::Document d;
|
||||
|
||||
char *buffer = (char *)std::malloc(p.size() + 1);
|
||||
std::memcpy(buffer, p.data(), p.size());
|
||||
buffer[p.size()] = '\0';
|
||||
#ifndef ALLPARSER
|
||||
if (!just_data)
|
||||
#endif
|
||||
{
|
||||
std::memcpy(buffer, p.data(), p.size());
|
||||
BEST_TIME("RapidJSON ",
|
||||
d.Parse<kParseValidateEncodingFlag>((const char *)buffer)
|
||||
.HasParseError(),
|
||||
false, , repeat, volume, !just_data);
|
||||
}
|
||||
#ifndef ALLPARSER
|
||||
if (!just_data)
|
||||
#endif
|
||||
{
|
||||
std::memcpy(buffer, p.data(), p.size());
|
||||
BEST_TIME("RapidJSON (accurate number parsing) ",
|
||||
d.Parse<kParseValidateEncodingFlag | kParseFullPrecisionFlag>(
|
||||
(const char *)buffer)
|
||||
.HasParseError(),
|
||||
false, , repeat, volume, !just_data);
|
||||
}
|
||||
BEST_TIME(
|
||||
"RapidJSON (insitu)",
|
||||
d.ParseInsitu<kParseValidateEncodingFlag>(buffer).HasParseError(), false,
|
||||
std::memcpy(buffer, p.data(), p.size()) && (buffer[p.size()] = '\0'),
|
||||
repeat, volume, !just_data);
|
||||
BEST_TIME("RapidJSON (insitu, accurate number parsing)",
|
||||
d.ParseInsitu<kParseValidateEncodingFlag | kParseFullPrecisionFlag>(
|
||||
buffer)
|
||||
.HasParseError(),
|
||||
false,
|
||||
std::memcpy(buffer, p.data(), p.size()) &&
|
||||
(buffer[p.size()] = '\0'),
|
||||
repeat, volume, !just_data);
|
||||
#ifdef HAS_BOOST_JSON
|
||||
{
|
||||
const boost::json::string_view sv(p.data(), p.size());
|
||||
boost::json::parser p;
|
||||
auto execute = [&p](auto sv) -> bool {
|
||||
boost::json::error_code ec;
|
||||
boost::json::monotonic_resource mr;
|
||||
p.reset( &mr );
|
||||
p.write(sv,ec);
|
||||
if(!ec)
|
||||
auto jv=p.release();
|
||||
return !!ec;
|
||||
};
|
||||
|
||||
BEST_TIME("Boost.json", execute(sv), false, , repeat, volume, !just_data);
|
||||
}
|
||||
#endif
|
||||
{
|
||||
|
||||
auto execute = [&p]() -> bool {
|
||||
yyjson_doc *doc = yyjson_read(p.data(), p.size(), 0);
|
||||
bool is_ok = doc != nullptr;
|
||||
yyjson_doc_free(doc);
|
||||
return is_ok;
|
||||
};
|
||||
|
||||
BEST_TIME("yyjson", execute(), true, , repeat, volume, !just_data);
|
||||
}
|
||||
#ifndef ALLPARSER
|
||||
if (!just_data)
|
||||
#endif
|
||||
BEST_TIME("sajson (dynamic mem)",
|
||||
sajson::parse(sajson::dynamic_allocation(),
|
||||
sajson::mutable_string_view(p.size(), buffer))
|
||||
.is_valid(),
|
||||
true, std::memcpy(buffer, p.data(), p.size()), repeat, volume,
|
||||
!just_data);
|
||||
|
||||
size_t ast_buffer_size = p.size();
|
||||
size_t *ast_buffer = (size_t *)std::malloc(ast_buffer_size * sizeof(size_t));
|
||||
// (static alloc, insitu)
|
||||
BEST_TIME(
|
||||
"sajson",
|
||||
sajson::parse(sajson::bounded_allocation(ast_buffer, ast_buffer_size),
|
||||
sajson::mutable_string_view(p.size(), buffer))
|
||||
.is_valid(),
|
||||
true, std::memcpy(buffer, p.data(), p.size()), repeat, volume,
|
||||
!just_data);
|
||||
|
||||
std::memcpy(buffer, p.data(), p.size());
|
||||
size_t expected = json::parse(p.data(), p.data() + p.size()).size();
|
||||
BEST_TIME("nlohmann-json", json::parse(buffer, buffer + p.size()).size(),
|
||||
expected, , repeat, volume, !just_data);
|
||||
|
||||
#ifdef ALLPARSER
|
||||
std::string json11err;
|
||||
BEST_TIME("dropbox (json11) ",
|
||||
((json11::Json::parse(buffer, json11err).is_null()) ||
|
||||
(!json11err.empty())),
|
||||
false, std::memcpy(buffer, p.data(), p.size()), repeat, volume,
|
||||
!just_data);
|
||||
|
||||
BEST_TIME("fastjson ", fastjson_parse(buffer), true,
|
||||
std::memcpy(buffer, p.data(), p.size()), repeat, volume,
|
||||
!just_data);
|
||||
JsonValue value;
|
||||
JsonAllocator allocator;
|
||||
char *endptr;
|
||||
BEST_TIME("gason ", jsonParse(buffer, &endptr, &value, allocator),
|
||||
JSON_OK, std::memcpy(buffer, p.data(), p.size()), repeat, volume,
|
||||
!just_data);
|
||||
void *state;
|
||||
BEST_TIME("ultrajson ",
|
||||
(UJDecode(buffer, p.size(), NULL, &state) == NULL), false,
|
||||
std::memcpy(buffer, p.data(), p.size()), repeat, volume,
|
||||
!just_data);
|
||||
|
||||
{
|
||||
std::unique_ptr<jsmntok_t[]> tokens =
|
||||
std::make_unique<jsmntok_t[]>(p.size());
|
||||
jsmn_parser jparser;
|
||||
jsmn_init(&jparser);
|
||||
std::memcpy(buffer, p.data(), p.size());
|
||||
buffer[p.size()] = '\0';
|
||||
BEST_TIME("jsmn ",
|
||||
(jsmn_parse(&jparser, buffer, p.size(), tokens.get(),
|
||||
static_cast<unsigned int>(p.size())) > 0),
|
||||
true, jsmn_init(&jparser), repeat, volume, !just_data);
|
||||
}
|
||||
std::memcpy(buffer, p.data(), p.size());
|
||||
buffer[p.size()] = '\0';
|
||||
cJSON *tree = cJSON_Parse(buffer);
|
||||
BEST_TIME("cJSON ", ((tree = cJSON_Parse(buffer)) != NULL), true,
|
||||
cJSON_Delete(tree), repeat, volume, !just_data);
|
||||
cJSON_Delete(tree);
|
||||
|
||||
Json::CharReaderBuilder b;
|
||||
Json::CharReader *json_cpp_reader = b.newCharReader();
|
||||
Json::Value root;
|
||||
Json::String errs;
|
||||
BEST_TIME("jsoncpp ",
|
||||
json_cpp_reader->parse(buffer, buffer + volume, &root, &errs), true,
|
||||
, repeat, volume, !just_data);
|
||||
delete json_cpp_reader;
|
||||
#endif
|
||||
if (!just_data)
|
||||
BEST_TIME("memcpy ",
|
||||
(std::memcpy(buffer, p.data(), p.size()) == buffer), true, ,
|
||||
repeat, volume, !just_data);
|
||||
#ifdef __linux__
|
||||
if (!just_data) {
|
||||
std::printf(
|
||||
"\n \n <doing additional analysis with performance counters (Linux "
|
||||
"only)>\n");
|
||||
std::vector<int> evts;
|
||||
evts.push_back(PERF_COUNT_HW_CPU_CYCLES);
|
||||
evts.push_back(PERF_COUNT_HW_INSTRUCTIONS);
|
||||
evts.push_back(PERF_COUNT_HW_BRANCH_MISSES);
|
||||
evts.push_back(PERF_COUNT_HW_CACHE_REFERENCES);
|
||||
evts.push_back(PERF_COUNT_HW_CACHE_MISSES);
|
||||
LinuxEvents<PERF_TYPE_HARDWARE> unified(evts);
|
||||
std::vector<unsigned long long> results;
|
||||
std::vector<unsigned long long> stats;
|
||||
results.resize(evts.size());
|
||||
stats.resize(evts.size());
|
||||
std::fill(stats.begin(), stats.end(), 0); // unnecessary
|
||||
for (decltype(repeat) i = 0; i < repeat; i++) {
|
||||
unified.start();
|
||||
auto parse_error = parser.parse(p).error();
|
||||
if (parse_error)
|
||||
std::printf("bug\n");
|
||||
unified.end(results);
|
||||
std::transform(stats.begin(), stats.end(), results.begin(), stats.begin(),
|
||||
std::plus<unsigned long long>());
|
||||
}
|
||||
std::printf(
|
||||
"simdjson : cycles %10.0f instructions %10.0f branchmisses %10.0f "
|
||||
"cacheref %10.0f cachemisses %10.0f bytespercachemiss %10.0f "
|
||||
"inspercycle %10.1f insperbyte %10.1f\n",
|
||||
static_cast<double>(stats[0]) / static_cast<double>(repeat),
|
||||
static_cast<double>(stats[1]) / static_cast<double>(repeat),
|
||||
static_cast<double>(stats[2]) / static_cast<double>(repeat),
|
||||
static_cast<double>(stats[3]) / static_cast<double>(repeat),
|
||||
static_cast<double>(stats[4]) / static_cast<double>(repeat),
|
||||
static_cast<double>(volume) * static_cast<double>(repeat) /
|
||||
static_cast<double>(stats[2]),
|
||||
static_cast<double>(stats[1]) / static_cast<double>(stats[0]),
|
||||
static_cast<double>(stats[1]) /
|
||||
(static_cast<double>(volume) * static_cast<double>(repeat)));
|
||||
|
||||
std::fill(stats.begin(), stats.end(), 0);
|
||||
for (decltype(repeat) i = 0; i < repeat; i++) {
|
||||
std::memcpy(buffer, p.data(), p.size());
|
||||
buffer[p.size()] = '\0';
|
||||
unified.start();
|
||||
if (d.ParseInsitu<kParseValidateEncodingFlag>(buffer).HasParseError() !=
|
||||
false)
|
||||
std::printf("bug\n");
|
||||
unified.end(results);
|
||||
std::transform(stats.begin(), stats.end(), results.begin(), stats.begin(),
|
||||
std::plus<unsigned long long>());
|
||||
}
|
||||
std::printf(
|
||||
"RapidJSON: cycles %10.0f instructions %10.0f branchmisses %10.0f "
|
||||
"cacheref %10.0f cachemisses %10.0f bytespercachemiss %10.0f "
|
||||
"inspercycle %10.1f insperbyte %10.1f\n",
|
||||
static_cast<double>(stats[0]) / static_cast<double>(repeat),
|
||||
static_cast<double>(stats[1]) / static_cast<double>(repeat),
|
||||
static_cast<double>(stats[2]) / static_cast<double>(repeat),
|
||||
static_cast<double>(stats[3]) / static_cast<double>(repeat),
|
||||
static_cast<double>(stats[4]) / static_cast<double>(repeat),
|
||||
static_cast<double>(volume) * static_cast<double>(repeat) /
|
||||
static_cast<double>(stats[2]),
|
||||
static_cast<double>(stats[1]) / static_cast<double>(stats[0]),
|
||||
static_cast<double>(stats[1]) /
|
||||
(static_cast<double>(volume) * static_cast<double>(repeat)));
|
||||
|
||||
std::fill(stats.begin(), stats.end(), 0); // unnecessary
|
||||
for (decltype(repeat) i = 0; i < repeat; i++) {
|
||||
std::memcpy(buffer, p.data(), p.size());
|
||||
unified.start();
|
||||
if (sajson::parse(sajson::bounded_allocation(ast_buffer, ast_buffer_size),
|
||||
sajson::mutable_string_view(p.size(), buffer))
|
||||
.is_valid() != true)
|
||||
std::printf("bug\n");
|
||||
unified.end(results);
|
||||
std::transform(stats.begin(), stats.end(), results.begin(), stats.begin(),
|
||||
std::plus<unsigned long long>());
|
||||
}
|
||||
std::printf(
|
||||
"sajson : cycles %10.0f instructions %10.0f branchmisses %10.0f "
|
||||
"cacheref %10.0f cachemisses %10.0f bytespercachemiss %10.0f "
|
||||
"inspercycle %10.1f insperbyte %10.1f\n",
|
||||
static_cast<double>(stats[0]) / static_cast<double>(repeat),
|
||||
static_cast<double>(stats[1]) / static_cast<double>(repeat),
|
||||
static_cast<double>(stats[2]) / static_cast<double>(repeat),
|
||||
static_cast<double>(stats[3]) / static_cast<double>(repeat),
|
||||
static_cast<double>(stats[4]) / static_cast<double>(repeat),
|
||||
static_cast<double>(volume) * static_cast<double>(repeat) /
|
||||
static_cast<double>(stats[2]),
|
||||
static_cast<double>(stats[1]) / static_cast<double>(stats[0]),
|
||||
static_cast<double>(stats[1]) /
|
||||
(static_cast<double>(volume) * static_cast<double>(repeat)));
|
||||
}
|
||||
#endif // __linux__
|
||||
|
||||
std::free(ast_buffer);
|
||||
std::free(buffer);
|
||||
return true;
|
||||
}
|
||||
|
||||
int main(int argc, char *argv[]) {
|
||||
bool verbose = false;
|
||||
bool justdata = false;
|
||||
bool just_data = false;
|
||||
double repeat_multiplier = 1;
|
||||
int c;
|
||||
while ((c = getopt(argc, argv, "vt")) != -1)
|
||||
while ((c = getopt(argc, argv, "r:vt")) != -1)
|
||||
switch (c) {
|
||||
case 'r':
|
||||
repeat_multiplier = atof(optarg);
|
||||
break;
|
||||
case 't':
|
||||
justdata = true;
|
||||
just_data = true;
|
||||
break;
|
||||
case 'v':
|
||||
verbose = true;
|
||||
@@ -77,190 +410,20 @@ int main(int argc, char *argv[]) {
|
||||
abort();
|
||||
}
|
||||
if (optind >= argc) {
|
||||
cerr << "Usage: " << argv[0] << " <jsonfile>\n";
|
||||
cerr << "Or " << argv[0] << " -v <jsonfile>\n";
|
||||
cerr << "To enable parsers that are not standard compliant, use the -a "
|
||||
"flag\n";
|
||||
std::cerr << "Usage: " << argv[0] << " <jsonfile>" << std::endl;
|
||||
std::cerr << "Or " << argv[0] << " -v <jsonfile>" << std::endl;
|
||||
std::cerr << "The '-t' flag outputs a table." << std::endl;
|
||||
std::cerr << "The '-r <N>' flag sets the repeat multiplier: set it above 1 "
|
||||
"to do more iterations, and below 1 to do fewer."
|
||||
<< std::endl;
|
||||
exit(1);
|
||||
}
|
||||
const char *filename = argv[optind];
|
||||
if (optind + 1 < argc) {
|
||||
cerr << "warning: ignoring everything after " << argv[optind + 1] << endl;
|
||||
int result = EXIT_SUCCESS;
|
||||
for (int fileind = optind; fileind < argc; fileind++) {
|
||||
if (!bench(argv[fileind], verbose, just_data, repeat_multiplier)) {
|
||||
result = EXIT_FAILURE;
|
||||
}
|
||||
std::printf("\n\n");
|
||||
}
|
||||
std::string_view p;
|
||||
try {
|
||||
p = get_corpus(filename);
|
||||
} catch (const std::exception &e) { // caught by reference to base
|
||||
std::cout << "Could not load the file " << filename << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
|
||||
if (verbose) {
|
||||
std::cout << "Input has ";
|
||||
if (p.size() > 1024 * 1024)
|
||||
std::cout << p.size() / (1024 * 1024) << " MB ";
|
||||
else if (p.size() > 1024)
|
||||
std::cout << p.size() / 1024 << " KB ";
|
||||
else
|
||||
std::cout << p.size() << " B ";
|
||||
std::cout << std::endl;
|
||||
}
|
||||
ParsedJson pj;
|
||||
bool allocok = pj.allocateCapacity(p.size(), 1024);
|
||||
|
||||
if (!allocok) {
|
||||
std::cerr << "can't allocate memory" << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
int repeat = 50;
|
||||
int volume = p.size();
|
||||
if(justdata) {
|
||||
printf("name cycles_per_byte cycles_per_byte_err gb_per_s gb_per_s_err \n");
|
||||
}
|
||||
if(!justdata) BEST_TIME("simdjson (dynamic mem) ", build_parsed_json(p).isValid(), true, ,
|
||||
repeat, volume, !justdata);
|
||||
// (static alloc)
|
||||
BEST_TIME("simdjson ", json_parse(p, pj), true, , repeat,
|
||||
volume, !justdata);
|
||||
|
||||
|
||||
rapidjson::Document d;
|
||||
|
||||
char *buffer = (char *)malloc(p.size() + 1);
|
||||
memcpy(buffer, p.data(), p.size());
|
||||
buffer[p.size()] = '\0';
|
||||
#ifndef ALLPARSER
|
||||
if(!justdata)
|
||||
#endif
|
||||
BEST_TIME(
|
||||
"RapidJSON ",
|
||||
d.Parse<kParseValidateEncodingFlag>((const char *)buffer).HasParseError(),
|
||||
false, memcpy(buffer, p.data(), p.size()), repeat, volume, !justdata);
|
||||
BEST_TIME("RapidJSON (insitu)",
|
||||
d.ParseInsitu<kParseValidateEncodingFlag>(buffer).HasParseError(),
|
||||
false, memcpy(buffer, p.data(), p.size()) && (buffer[p.size()] = '\0'), repeat, volume, !justdata);
|
||||
#ifndef ALLPARSER
|
||||
if(!justdata)
|
||||
#endif
|
||||
BEST_TIME("sajson (dynamic mem)",
|
||||
sajson::parse(sajson::dynamic_allocation(),
|
||||
sajson::mutable_string_view(p.size(), buffer))
|
||||
.is_valid(),
|
||||
true, memcpy(buffer, p.data(), p.size()), repeat, volume, !justdata);
|
||||
|
||||
size_t astbuffersize = p.size();
|
||||
size_t *ast_buffer = (size_t *)malloc(astbuffersize * sizeof(size_t));
|
||||
// (static alloc, insitu)
|
||||
BEST_TIME("sajson",
|
||||
sajson::parse(sajson::bounded_allocation(ast_buffer, astbuffersize),
|
||||
sajson::mutable_string_view(p.size(), buffer))
|
||||
.is_valid(),
|
||||
true, memcpy(buffer, p.data(), p.size()), repeat, volume, !justdata);
|
||||
#ifdef __linux__
|
||||
if(!justdata) {
|
||||
vector<int> evts;
|
||||
evts.push_back(PERF_COUNT_HW_CPU_CYCLES);
|
||||
evts.push_back(PERF_COUNT_HW_INSTRUCTIONS);
|
||||
evts.push_back(PERF_COUNT_HW_BRANCH_MISSES);
|
||||
evts.push_back(PERF_COUNT_HW_CACHE_REFERENCES);
|
||||
evts.push_back(PERF_COUNT_HW_CACHE_MISSES);
|
||||
LinuxEvents<PERF_TYPE_HARDWARE> unified(evts);
|
||||
vector<unsigned long long> results;
|
||||
vector<unsigned long long> stats;
|
||||
results.resize(evts.size());
|
||||
stats.resize(evts.size());
|
||||
std::fill(stats.begin(), stats.end(), 0);// unnecessary
|
||||
for(int i = 0; i < repeat; i++) {
|
||||
unified.start();
|
||||
if(json_parse(p, pj) != true) printf("bug\n");
|
||||
unified.end(results);
|
||||
std::transform (stats.begin(), stats.end(), results.begin(), stats.begin(), std::plus<unsigned long long>());
|
||||
}
|
||||
printf("simdjson : cycles %10.0f instructions %10.0f branchmisses %10.0f cacheref %10.0f cachemisses %10.0f bytespercachemiss %10.0f inspercycle %10.1f insperbyte %10.1f\n",
|
||||
stats[0] * 1.0 / repeat, stats[1] * 1.0 / repeat, stats[2] * 1.0 / repeat, stats[3] * 1.0 / repeat, stats[4] * 1.0 / repeat, volume * repeat * 1.0 / stats[2], stats[1] *1.0 / stats[0], stats[1] * 1.0 / (volume * repeat));
|
||||
|
||||
std::fill(stats.begin(), stats.end(), 0);
|
||||
for(int i = 0; i < repeat; i++) {
|
||||
memcpy(buffer, p.data(), p.size());
|
||||
buffer[p.size()] = '\0';
|
||||
unified.start();
|
||||
if(d.ParseInsitu<kParseValidateEncodingFlag>(buffer).HasParseError() != false) printf("bug\n");
|
||||
unified.end(results);
|
||||
std::transform (stats.begin(), stats.end(), results.begin(), stats.begin(), std::plus<unsigned long long>());
|
||||
}
|
||||
printf("RapidJSON: cycles %10.0f instructions %10.0f branchmisses %10.0f cacheref %10.0f cachemisses %10.0f bytespercachemiss %10.0f inspercycle %10.1f insperbyte %10.1f\n",
|
||||
stats[0] * 1.0 / repeat, stats[1] * 1.0 / repeat, stats[2] * 1.0 / repeat, stats[3] * 1.0 / repeat, stats[4] * 1.0 / repeat, volume * repeat * 1.0 / stats[2], stats[1] *1.0 / stats[0], stats[1] * 1.0 / (volume * repeat));
|
||||
|
||||
std::fill(stats.begin(), stats.end(), 0);// unnecessary
|
||||
for(int i = 0; i < repeat; i++) {
|
||||
memcpy(buffer, p.data(), p.size());
|
||||
unified.start();
|
||||
if(sajson::parse(sajson::bounded_allocation(ast_buffer, astbuffersize),
|
||||
sajson::mutable_string_view(p.size(), buffer))
|
||||
.is_valid() != true) printf("bug\n");
|
||||
unified.end(results);
|
||||
std::transform (stats.begin(), stats.end(), results.begin(), stats.begin(), std::plus<unsigned long long>());
|
||||
}
|
||||
printf("sajson : cycles %10.0f instructions %10.0f branchmisses %10.0f cacheref %10.0f cachemisses %10.0f bytespercachemiss %10.0f inspercycle %10.1f insperbyte %10.1f\n",
|
||||
stats[0] * 1.0 / repeat, stats[1] * 1.0 / repeat, stats[2] * 1.0 / repeat, stats[3] * 1.0 / repeat, stats[4] * 1.0 / repeat, volume * repeat * 1.0 / stats[2], stats[1] *1.0 / stats[0], stats[1] * 1.0 / (volume * repeat));
|
||||
}
|
||||
#endif// __linux__
|
||||
#ifdef ALLPARSER
|
||||
std::string json11err;
|
||||
BEST_TIME("dropbox (json11) ",
|
||||
((json11::Json::parse(buffer, json11err).is_null()) ||
|
||||
(!json11err.empty())),
|
||||
false, memcpy(buffer, p.data(), p.size()), repeat, volume, !justdata);
|
||||
|
||||
BEST_TIME("fastjson ", fastjson_parse(buffer), true,
|
||||
memcpy(buffer, p.data(), p.size()), repeat, volume, !justdata);
|
||||
JsonValue value;
|
||||
JsonAllocator allocator;
|
||||
char *endptr;
|
||||
BEST_TIME("gason ",
|
||||
jsonParse(buffer, &endptr, &value, allocator), JSON_OK,
|
||||
memcpy(buffer, p.data(), p.size()), repeat, volume, !justdata);
|
||||
void *state;
|
||||
BEST_TIME("ultrajson ",
|
||||
(UJDecode(buffer, p.size(), NULL, &state) == NULL), false,
|
||||
memcpy(buffer, p.data(), p.size()), repeat, volume, !justdata);
|
||||
|
||||
|
||||
|
||||
auto * tokens = make_unique<jsmntok_t[](p.size());
|
||||
if(tokens == NULL) {
|
||||
printf("Failed to alloc memory for jsmn\n");
|
||||
} else {
|
||||
jsmn_parser parser;
|
||||
jsmn_init(&parser);
|
||||
memcpy(buffer, p.data(), p.size());
|
||||
buffer[p.size()] = '\0';
|
||||
BEST_TIME("jsmn ",
|
||||
(jsmn_parse(&parser, buffer, p.size(), tokens.get(), p.size()) > 0), true,
|
||||
jsmn_init(&parser), repeat, volume, !justdata);
|
||||
}
|
||||
|
||||
memcpy(buffer, p.data(), p.size());
|
||||
buffer[p.size()] = '\0';
|
||||
cJSON * tree = cJSON_Parse(buffer);
|
||||
BEST_TIME("cJSON ",
|
||||
((tree = cJSON_Parse(buffer)) != NULL ), true,
|
||||
cJSON_Delete(tree), repeat, volume, !justdata);
|
||||
cJSON_Delete(tree);
|
||||
|
||||
Json::CharReaderBuilder b;
|
||||
Json::CharReader * jsoncppreader = b.newCharReader();
|
||||
Json::Value root;
|
||||
Json::String errs;
|
||||
BEST_TIME("jsoncpp ",
|
||||
jsoncppreader->parse(buffer,buffer+volume,&root,&errs), true,
|
||||
, repeat, volume, !justdata);
|
||||
delete jsoncppreader;
|
||||
#endif
|
||||
if(!justdata) BEST_TIME("memcpy ",
|
||||
(memcpy(buffer, p.data(), p.size()) == buffer), true, , repeat,
|
||||
volume, !justdata);
|
||||
aligned_free((void *)p.data());
|
||||
free(ast_buffer);
|
||||
free(buffer);
|
||||
return result;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,51 @@
|
||||
#pragma once
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
|
||||
#include "partial_tweets.h"
|
||||
|
||||
namespace partial_tweets {
|
||||
|
||||
using namespace simdjson;
|
||||
|
||||
class Dom {
|
||||
public:
|
||||
simdjson_really_inline bool Run(const padded_string &json);
|
||||
|
||||
simdjson_really_inline const std::vector<tweet> &Result() { return tweets; }
|
||||
simdjson_really_inline size_t ItemCount() { return tweets.size(); }
|
||||
|
||||
private:
|
||||
dom::parser parser{};
|
||||
std::vector<tweet> tweets{};
|
||||
|
||||
simdjson_really_inline uint64_t nullable_int(dom::element element) {
|
||||
if (element.is_null()) { return 0; }
|
||||
return element;
|
||||
}
|
||||
};
|
||||
|
||||
simdjson_really_inline bool Dom::Run(const padded_string &json) {
|
||||
tweets.clear();
|
||||
|
||||
for (dom::element tweet : parser.parse(json)["statuses"]) {
|
||||
auto user = tweet["user"];
|
||||
tweets.emplace_back(partial_tweets::tweet{
|
||||
tweet["created_at"],
|
||||
tweet["id"],
|
||||
tweet["text"],
|
||||
nullable_int(tweet["in_reply_to_status_id"]),
|
||||
{ user["id"], user["screen_name"] },
|
||||
tweet["retweet_count"],
|
||||
tweet["favorite_count"]
|
||||
});
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
BENCHMARK_TEMPLATE(PartialTweets, Dom);
|
||||
|
||||
} // namespace partial_tweets
|
||||
|
||||
#endif // SIMDJSON_EXCEPTIONS
|
||||
@@ -0,0 +1,64 @@
|
||||
#pragma once
|
||||
|
||||
#include "partial_tweets.h"
|
||||
|
||||
namespace partial_tweets {
|
||||
|
||||
using namespace simdjson;
|
||||
|
||||
class DomNoExcept {
|
||||
public:
|
||||
simdjson_really_inline bool Run(const simdjson::padded_string &json) noexcept;
|
||||
|
||||
simdjson_really_inline const std::vector<tweet> &Result() { return tweets; }
|
||||
simdjson_really_inline size_t ItemCount() { return tweets.size(); }
|
||||
|
||||
private:
|
||||
dom::parser parser{};
|
||||
std::vector<tweet> tweets{};
|
||||
|
||||
simdjson_really_inline simdjson_result<uint64_t> nullable_int(simdjson_result<dom::element> result) noexcept {
|
||||
dom::element element;
|
||||
SIMDJSON_TRY( result.get(element) );
|
||||
if (element.is_null()) { return 0; }
|
||||
return element.get_uint64();
|
||||
}
|
||||
|
||||
simdjson_really_inline error_code RunNoExcept(const simdjson::padded_string &json) noexcept;
|
||||
};
|
||||
|
||||
simdjson_really_inline bool DomNoExcept::Run(const simdjson::padded_string &json) noexcept {
|
||||
auto error = RunNoExcept(json);
|
||||
if (error) { std::cerr << error << std::endl; return false; }
|
||||
return true;
|
||||
}
|
||||
|
||||
simdjson_really_inline error_code DomNoExcept::RunNoExcept(const simdjson::padded_string &json) noexcept {
|
||||
tweets.clear();
|
||||
|
||||
dom::array tweet_array;
|
||||
SIMDJSON_TRY( parser.parse(json)["statuses"].get_array().get(tweet_array) );
|
||||
|
||||
for (auto tweet_element : tweet_array) {
|
||||
dom::object tweet;
|
||||
SIMDJSON_TRY( tweet_element.get_object().get(tweet) );
|
||||
|
||||
dom::object user;
|
||||
SIMDJSON_TRY( tweet["user"].get_object().get(user) );
|
||||
|
||||
partial_tweets::tweet t;
|
||||
SIMDJSON_TRY( tweet["created_at"] .get_string().get(t.created_at) );
|
||||
SIMDJSON_TRY( tweet["id"] .get_uint64().get(t.id) );
|
||||
SIMDJSON_TRY( tweet["text"] .get_string().get(t.text) );
|
||||
SIMDJSON_TRY( nullable_int(tweet["in_reply_to_status_id"]).get(t.in_reply_to_status_id) );
|
||||
SIMDJSON_TRY( user["id"] .get_uint64().get(t.user.id) );
|
||||
SIMDJSON_TRY( user["screen_name"] .get_string().get(t.user.screen_name) );
|
||||
SIMDJSON_TRY( tweet["retweet_count"] .get_uint64().get(t.retweet_count) );
|
||||
SIMDJSON_TRY( tweet["favorite_count"].get_uint64().get(t.favorite_count) );
|
||||
|
||||
tweets.push_back(t);
|
||||
}
|
||||
return SUCCESS;
|
||||
}
|
||||
|
||||
} // namespace partial_tweets
|
||||
@@ -0,0 +1,93 @@
|
||||
#pragma once
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
|
||||
#include "partial_tweets.h"
|
||||
|
||||
namespace partial_tweets {
|
||||
|
||||
using namespace simdjson;
|
||||
using namespace simdjson::builtin;
|
||||
|
||||
class Iter {
|
||||
public:
|
||||
simdjson_really_inline bool Run(const padded_string &json);
|
||||
|
||||
simdjson_really_inline const std::vector<tweet> &Result() { return tweets; }
|
||||
simdjson_really_inline size_t ItemCount() { return tweets.size(); }
|
||||
|
||||
private:
|
||||
ondemand::parser parser{};
|
||||
std::vector<tweet> tweets{};
|
||||
|
||||
simdjson_really_inline uint64_t nullable_int(ondemand::value && value) {
|
||||
if (value.is_null()) { return 0; }
|
||||
return std::move(value);
|
||||
}
|
||||
|
||||
simdjson_really_inline twitter_user read_user(ondemand::object && user) {
|
||||
// Move user into a local object so it gets destroyed (and moves the iterator)
|
||||
ondemand::object u = std::move(user);
|
||||
return { u["id"], u["screen_name"] };
|
||||
}
|
||||
};
|
||||
|
||||
simdjson_really_inline bool Iter::Run(const padded_string &json) {
|
||||
tweets.clear();
|
||||
|
||||
// Walk the document, parsing the tweets as we go
|
||||
|
||||
// { "statuses":
|
||||
auto iter = parser.iterate_raw(json).value();
|
||||
if (!iter.start_object() || !iter.find_field_raw("statuses")) { return false; }
|
||||
// { "statuses": [
|
||||
if (!iter.start_array()) { return false; }
|
||||
|
||||
do {
|
||||
tweet tweet;
|
||||
|
||||
if (!iter.start_object() || !iter.find_field_raw("created_at")) { return false; }
|
||||
tweet.created_at = iter.consume_string();
|
||||
|
||||
if (!iter.has_next_field() || !iter.find_field_raw("id")) { return false; }
|
||||
tweet.id = iter.consume_uint64();
|
||||
|
||||
if (!iter.has_next_field() || !iter.find_field_raw("text")) { return false; }
|
||||
tweet.text = iter.consume_string();
|
||||
|
||||
if (!iter.has_next_field() || !iter.find_field_raw("in_reply_to_status_id")) { return false; }
|
||||
if (!iter.is_null()) {
|
||||
tweet.in_reply_to_status_id = iter.consume_uint64();
|
||||
}
|
||||
|
||||
if (!iter.has_next_field() || !iter.find_field_raw("user")) { return false; }
|
||||
{
|
||||
if (!iter.start_object() || !iter.find_field_raw("id")) { return false; }
|
||||
tweet.user.id = iter.consume_uint64();
|
||||
|
||||
if (!iter.has_next_field() || !iter.find_field_raw("screen_name")) { return false; }
|
||||
tweet.user.screen_name = iter.consume_string();
|
||||
|
||||
if (iter.skip_container()) { return false; } // Skip the rest of the user object
|
||||
}
|
||||
|
||||
if (!iter.has_next_field() || !iter.find_field_raw("retweet_count")) { return false; }
|
||||
tweet.retweet_count = iter.consume_uint64();
|
||||
|
||||
if (!iter.has_next_field() || !iter.find_field_raw("favorite_count")) { return false; }
|
||||
tweet.favorite_count = iter.consume_uint64();
|
||||
|
||||
tweets.push_back(tweet);
|
||||
|
||||
if (iter.skip_container()) { return false; } // Skip the rest of the tweet object
|
||||
|
||||
} while (iter.has_next_element());
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
BENCHMARK_TEMPLATE(PartialTweets, Iter);
|
||||
|
||||
} // namespace partial_tweets
|
||||
|
||||
#endif // SIMDJSON_EXCEPTIONS
|
||||
@@ -0,0 +1,65 @@
|
||||
#pragma once
|
||||
|
||||
#if SIMDJSON_EXCEPTIONS
|
||||
|
||||
#include "partial_tweets.h"
|
||||
|
||||
namespace partial_tweets {
|
||||
|
||||
using namespace simdjson;
|
||||
using namespace simdjson::builtin;
|
||||
|
||||
|
||||
class OnDemand {
|
||||
public:
|
||||
OnDemand() {
|
||||
if(!displayed_implementation) {
|
||||
std::cout << "On Demand implementation: " << builtin_implementation()->name() << std::endl;
|
||||
displayed_implementation = true;
|
||||
}
|
||||
}
|
||||
simdjson_really_inline bool Run(const padded_string &json);
|
||||
simdjson_really_inline const std::vector<tweet> &Result() { return tweets; }
|
||||
simdjson_really_inline size_t ItemCount() { return tweets.size(); }
|
||||
|
||||
private:
|
||||
ondemand::parser parser{};
|
||||
std::vector<tweet> tweets{};
|
||||
|
||||
simdjson_really_inline uint64_t nullable_int(ondemand::value && value) {
|
||||
if (value.is_null()) { return 0; }
|
||||
return std::move(value);
|
||||
}
|
||||
|
||||
simdjson_really_inline twitter_user read_user(ondemand::object && user) {
|
||||
// Move user into a local object so it gets destroyed (and moves the iterator)
|
||||
ondemand::object u = std::move(user);
|
||||
return { u["id"], u["screen_name"] };
|
||||
}
|
||||
static inline bool displayed_implementation = false;
|
||||
};
|
||||
|
||||
simdjson_really_inline bool OnDemand::Run(const padded_string &json) {
|
||||
tweets.clear();
|
||||
|
||||
// Walk the document, parsing the tweets as we go
|
||||
auto doc = parser.iterate(json);
|
||||
for (ondemand::object tweet : doc["statuses"]) {
|
||||
tweets.emplace_back(partial_tweets::tweet{
|
||||
tweet["created_at"],
|
||||
tweet["id"],
|
||||
tweet["text"],
|
||||
nullable_int(tweet["in_reply_to_status_id"]),
|
||||
read_user(tweet["user"]),
|
||||
tweet["retweet_count"],
|
||||
tweet["favorite_count"]
|
||||
});
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
BENCHMARK_TEMPLATE(PartialTweets, OnDemand);
|
||||
|
||||
} // namespace partial_tweets
|
||||
|
||||
#endif // SIMDJSON_EXCEPTIONS
|
||||
@@ -0,0 +1,41 @@
|
||||
#pragma once
|
||||
|
||||
//
|
||||
// Interface
|
||||
//
|
||||
|
||||
namespace partial_tweets {
|
||||
template<typename T> static void PartialTweets(benchmark::State &state);
|
||||
} // namespace partial_tweets
|
||||
|
||||
//
|
||||
// Implementation
|
||||
//
|
||||
|
||||
#include "tweet.h"
|
||||
#include <vector>
|
||||
#include "event_counter.h"
|
||||
#include "domnoexcept.h"
|
||||
#include "json_benchmark.h"
|
||||
|
||||
namespace partial_tweets {
|
||||
|
||||
using namespace simdjson;
|
||||
|
||||
template<typename T> static void PartialTweets(benchmark::State &state) {
|
||||
//
|
||||
// Load the JSON file
|
||||
//
|
||||
constexpr const char *TWITTER_JSON = SIMDJSON_BENCHMARK_DATA_DIR "twitter.json";
|
||||
error_code error;
|
||||
padded_string json;
|
||||
if ((error = padded_string::load(TWITTER_JSON).get(json))) {
|
||||
std::cerr << error << std::endl;
|
||||
state.SkipWithError("error loading");
|
||||
return;
|
||||
}
|
||||
|
||||
JsonBenchmark<T, DomNoExcept>(state, json);
|
||||
}
|
||||
|
||||
} // namespace partial_tweets
|
||||
@@ -0,0 +1,69 @@
|
||||
#pragma once
|
||||
|
||||
|
||||
#include "partial_tweets.h"
|
||||
#include "sax_tweet_reader_visitor.h"
|
||||
|
||||
namespace partial_tweets {
|
||||
|
||||
using namespace simdjson;
|
||||
using namespace simdjson::builtin;
|
||||
using namespace simdjson::builtin::stage2;
|
||||
|
||||
class Sax {
|
||||
public:
|
||||
simdjson_really_inline bool Run(const padded_string &json) noexcept;
|
||||
|
||||
simdjson_really_inline const std::vector<tweet> &Result() { return tweets; }
|
||||
simdjson_really_inline size_t ItemCount() { return tweets.size(); }
|
||||
|
||||
private:
|
||||
simdjson_really_inline error_code RunNoExcept(const padded_string &json) noexcept;
|
||||
error_code Allocate(size_t new_capacity);
|
||||
std::unique_ptr<uint8_t[]> string_buf{};
|
||||
size_t capacity{};
|
||||
dom_parser_implementation dom_parser{};
|
||||
std::vector<tweet> tweets{};
|
||||
};
|
||||
|
||||
// NOTE: this assumes the dom_parser is already allocated
|
||||
bool Sax::Run(const padded_string &json) noexcept {
|
||||
auto error = RunNoExcept(json);
|
||||
if (error) { std::cerr << error << std::endl; return false; }
|
||||
return true;
|
||||
}
|
||||
|
||||
error_code Sax::RunNoExcept(const padded_string &json) noexcept {
|
||||
tweets.clear();
|
||||
|
||||
// Allocate capacity if needed
|
||||
if (capacity < json.size()) {
|
||||
SIMDJSON_TRY( Allocate(json.size()) );
|
||||
}
|
||||
|
||||
// Run stage 1 first.
|
||||
SIMDJSON_TRY( dom_parser.stage1((uint8_t *)json.data(), json.size(), false) );
|
||||
|
||||
// Then walk the document, parsing the tweets as we go
|
||||
json_iterator iter(dom_parser, 0);
|
||||
sax_tweet_reader_visitor visitor(tweets, string_buf.get());
|
||||
SIMDJSON_TRY( iter.walk_document<false>(visitor) );
|
||||
return SUCCESS;
|
||||
}
|
||||
|
||||
error_code Sax::Allocate(size_t new_capacity) {
|
||||
// string_capacity copied from document::allocate
|
||||
size_t string_capacity = SIMDJSON_ROUNDUP_N(5 * new_capacity / 3 + SIMDJSON_PADDING, 64);
|
||||
string_buf.reset(new (std::nothrow) uint8_t[string_capacity]);
|
||||
if (auto error = dom_parser.set_capacity(new_capacity)) { return error; }
|
||||
if (capacity == 0) { // set max depth the first time only
|
||||
if (auto error = dom_parser.set_max_depth(DEFAULT_MAX_DEPTH)) { return error; }
|
||||
}
|
||||
capacity = new_capacity;
|
||||
return SUCCESS;
|
||||
}
|
||||
|
||||
BENCHMARK_TEMPLATE(PartialTweets, Sax);
|
||||
|
||||
} // namespace partial_tweets
|
||||
|
||||
@@ -0,0 +1,514 @@
|
||||
#pragma once
|
||||
|
||||
#include "simdjson.h"
|
||||
#include "tweet.h"
|
||||
#include <vector>
|
||||
|
||||
namespace partial_tweets {
|
||||
|
||||
using namespace simdjson;
|
||||
using namespace simdjson::builtin;
|
||||
using namespace simdjson::builtin::stage2;
|
||||
|
||||
struct sax_tweet_reader_visitor {
|
||||
public:
|
||||
simdjson_really_inline sax_tweet_reader_visitor(std::vector<tweet> &tweets, uint8_t *string_buf);
|
||||
|
||||
simdjson_really_inline error_code visit_document_start(json_iterator &iter);
|
||||
simdjson_really_inline error_code visit_object_start(json_iterator &iter);
|
||||
simdjson_really_inline error_code visit_key(json_iterator &iter, const uint8_t *key);
|
||||
simdjson_really_inline error_code visit_primitive(json_iterator &iter, const uint8_t *value);
|
||||
simdjson_really_inline error_code visit_array_start(json_iterator &iter);
|
||||
simdjson_really_inline error_code visit_array_end(json_iterator &iter);
|
||||
simdjson_really_inline error_code visit_object_end(json_iterator &iter);
|
||||
simdjson_really_inline error_code visit_document_end(json_iterator &iter);
|
||||
simdjson_really_inline error_code visit_empty_array(json_iterator &iter);
|
||||
simdjson_really_inline error_code visit_empty_object(json_iterator &iter);
|
||||
simdjson_really_inline error_code visit_root_primitive(json_iterator &iter, const uint8_t *value);
|
||||
simdjson_really_inline error_code increment_count(json_iterator &iter);
|
||||
|
||||
private:
|
||||
// Since we only care about one thing at each level, we just use depth as the marker for what
|
||||
// object/array we're nested inside.
|
||||
enum class containers {
|
||||
document = 0, //
|
||||
top_object = 1, // {
|
||||
statuses = 2, // { "statuses": [
|
||||
tweet = 3, // { "statuses": [ {
|
||||
user = 4 // { "statuses": [ { "user": {
|
||||
};
|
||||
/**
|
||||
* The largest depth we care about.
|
||||
* There can be things at lower depths.
|
||||
*/
|
||||
static constexpr uint32_t MAX_SUPPORTED_DEPTH = uint32_t(containers::user);
|
||||
static constexpr const char *STATE_NAMES[] = {
|
||||
"document",
|
||||
"top object",
|
||||
"statuses",
|
||||
"tweet",
|
||||
"user"
|
||||
};
|
||||
enum class field_type {
|
||||
any,
|
||||
unsigned_integer,
|
||||
string,
|
||||
nullable_unsigned_integer,
|
||||
object,
|
||||
array
|
||||
};
|
||||
struct field {
|
||||
const char * key{};
|
||||
size_t len{0};
|
||||
size_t offset;
|
||||
containers container{containers::document};
|
||||
field_type type{field_type::any};
|
||||
};
|
||||
|
||||
std::vector<tweet> &tweets;
|
||||
containers container{containers::document};
|
||||
uint8_t *current_string_buf_loc;
|
||||
const uint8_t *current_key{};
|
||||
|
||||
simdjson_really_inline bool in_container(json_iterator &iter);
|
||||
simdjson_really_inline bool in_container_child(json_iterator &iter);
|
||||
simdjson_really_inline void start_container(json_iterator &iter);
|
||||
simdjson_really_inline void end_container(json_iterator &iter);
|
||||
simdjson_really_inline error_code parse_nullable_unsigned(json_iterator &iter, const uint8_t *value, const field &f);
|
||||
simdjson_really_inline error_code parse_unsigned(json_iterator &iter, const uint8_t *value, const field &f);
|
||||
simdjson_really_inline error_code parse_string(json_iterator &iter, const uint8_t *value, const field &f);
|
||||
|
||||
struct field_lookup {
|
||||
field entries[256]{};
|
||||
|
||||
field_lookup();
|
||||
simdjson_really_inline field get(const uint8_t * key, containers container);
|
||||
private:
|
||||
simdjson_really_inline uint8_t hash(const char * key, uint32_t depth);
|
||||
simdjson_really_inline void add(const char * key, size_t len, containers container, field_type type, size_t offset);
|
||||
simdjson_really_inline void neg(const char * const key, uint32_t depth);
|
||||
};
|
||||
static field_lookup fields;
|
||||
}; // sax_tweet_reader_visitor
|
||||
|
||||
simdjson_really_inline sax_tweet_reader_visitor::sax_tweet_reader_visitor(std::vector<tweet> &_tweets, uint8_t *_string_buf)
|
||||
: tweets{_tweets},
|
||||
current_string_buf_loc{_string_buf} {
|
||||
}
|
||||
|
||||
simdjson_really_inline error_code sax_tweet_reader_visitor::visit_document_start(json_iterator &iter) {
|
||||
start_container(iter);
|
||||
return SUCCESS;
|
||||
}
|
||||
simdjson_really_inline error_code sax_tweet_reader_visitor::visit_array_start(json_iterator &iter) {
|
||||
// If we're not in a container we care about, don't bother with the rest
|
||||
if (!in_container_child(iter)) { return SUCCESS; }
|
||||
|
||||
// Handle fields first
|
||||
if (current_key) {
|
||||
switch (fields.get(current_key, container).type) {
|
||||
case field_type::array: // { "statuses": [
|
||||
start_container(iter);
|
||||
current_key = nullptr;
|
||||
return SUCCESS;
|
||||
case field_type::any:
|
||||
return SUCCESS;
|
||||
case field_type::object:
|
||||
case field_type::unsigned_integer:
|
||||
case field_type::nullable_unsigned_integer:
|
||||
case field_type::string:
|
||||
iter.log_error("unexpected array field");
|
||||
return INCORRECT_TYPE;
|
||||
}
|
||||
}
|
||||
|
||||
// We're not in a field, so it must be a child of an array. We support any of those.
|
||||
iter.log_error("unexpected array");
|
||||
return INCORRECT_TYPE;
|
||||
}
|
||||
simdjson_really_inline error_code sax_tweet_reader_visitor::visit_object_start(json_iterator &iter) {
|
||||
// If we're not in a container we care about, don't bother with the rest
|
||||
if (!in_container_child(iter)) { return SUCCESS; }
|
||||
|
||||
// Handle known fields
|
||||
if (current_key) {
|
||||
auto f = fields.get(current_key, container);
|
||||
switch (f.type) {
|
||||
case field_type::object: // { "statuses": [ { "user": {
|
||||
start_container(iter);
|
||||
return SUCCESS;
|
||||
case field_type::any:
|
||||
return SUCCESS;
|
||||
case field_type::array:
|
||||
case field_type::unsigned_integer:
|
||||
case field_type::nullable_unsigned_integer:
|
||||
case field_type::string:
|
||||
iter.log_error("unexpected object field");
|
||||
return INCORRECT_TYPE;
|
||||
}
|
||||
}
|
||||
|
||||
// It's not a field, so it's a child of an array or document
|
||||
switch (container) {
|
||||
case containers::document: // top_object: {
|
||||
case containers::statuses: // tweet: { "statuses": [ {
|
||||
start_container(iter);
|
||||
return SUCCESS;
|
||||
case containers::top_object:
|
||||
case containers::tweet:
|
||||
case containers::user:
|
||||
iter.log_error("unexpected object");
|
||||
return INCORRECT_TYPE;
|
||||
}
|
||||
SIMDJSON_UNREACHABLE();
|
||||
return UNINITIALIZED;
|
||||
}
|
||||
simdjson_really_inline error_code sax_tweet_reader_visitor::visit_key(json_iterator &, const uint8_t *key) {
|
||||
current_key = key;
|
||||
return SUCCESS;
|
||||
}
|
||||
simdjson_really_inline error_code sax_tweet_reader_visitor::visit_primitive(json_iterator &iter, const uint8_t *value) {
|
||||
// Don't bother unless we're in a container we care about
|
||||
if (!in_container(iter)) { return SUCCESS; }
|
||||
|
||||
// Handle fields first
|
||||
if (current_key) {
|
||||
auto f = fields.get(current_key, container);
|
||||
switch (f.type) {
|
||||
case field_type::unsigned_integer:
|
||||
return parse_unsigned(iter, value, f);
|
||||
case field_type::nullable_unsigned_integer:
|
||||
return parse_nullable_unsigned(iter, value, f);
|
||||
case field_type::string:
|
||||
return parse_string(iter, value, f);
|
||||
case field_type::any:
|
||||
return SUCCESS;
|
||||
case field_type::array:
|
||||
case field_type::object:
|
||||
iter.log_error("unexpected primitive");
|
||||
return INCORRECT_TYPE;
|
||||
}
|
||||
current_key = nullptr;
|
||||
}
|
||||
|
||||
// If it's not a field, it's a child of an array.
|
||||
// The only array we support is statuses, which must contain objects.
|
||||
iter.log_error("unexpected primitive");
|
||||
return INCORRECT_TYPE;
|
||||
}
|
||||
simdjson_really_inline error_code sax_tweet_reader_visitor::visit_array_end(json_iterator &iter) {
|
||||
if (in_container(iter)) { end_container(iter); }
|
||||
return SUCCESS;
|
||||
}
|
||||
simdjson_really_inline error_code sax_tweet_reader_visitor::visit_object_end(json_iterator &iter) {
|
||||
current_key = nullptr;
|
||||
if (in_container(iter)) { end_container(iter); }
|
||||
return SUCCESS;
|
||||
}
|
||||
|
||||
simdjson_really_inline error_code sax_tweet_reader_visitor::visit_document_end(json_iterator &) {
|
||||
return SUCCESS;
|
||||
}
|
||||
|
||||
simdjson_really_inline error_code sax_tweet_reader_visitor::visit_empty_array(json_iterator &) {
|
||||
current_key = nullptr;
|
||||
return SUCCESS;
|
||||
}
|
||||
simdjson_really_inline error_code sax_tweet_reader_visitor::visit_empty_object(json_iterator &) {
|
||||
return SUCCESS;
|
||||
}
|
||||
simdjson_really_inline error_code sax_tweet_reader_visitor::visit_root_primitive(json_iterator &iter, const uint8_t *) {
|
||||
iter.log_error("unexpected root primitive");
|
||||
return INCORRECT_TYPE;
|
||||
}
|
||||
|
||||
simdjson_really_inline error_code sax_tweet_reader_visitor::increment_count(json_iterator &) { return SUCCESS; }
|
||||
|
||||
simdjson_really_inline bool sax_tweet_reader_visitor::in_container(json_iterator &iter) {
|
||||
return iter.depth == uint32_t(container);
|
||||
}
|
||||
simdjson_really_inline bool sax_tweet_reader_visitor::in_container_child(json_iterator &iter) {
|
||||
return iter.depth == uint32_t(container) + 1;
|
||||
}
|
||||
simdjson_really_inline void sax_tweet_reader_visitor::start_container(json_iterator &iter) {
|
||||
SIMDJSON_ASSUME(iter.depth <= MAX_SUPPORTED_DEPTH); // Asserts in debug mode
|
||||
container = containers(iter.depth);
|
||||
if (logger::LOG_ENABLED) { iter.log_value(STATE_NAMES[iter.depth]); }
|
||||
if (container == containers::tweet) { tweets.push_back({}); }
|
||||
}
|
||||
simdjson_really_inline void sax_tweet_reader_visitor::end_container(json_iterator &) {
|
||||
container = containers(int(container) - 1);
|
||||
}
|
||||
simdjson_really_inline error_code sax_tweet_reader_visitor::parse_nullable_unsigned(json_iterator &iter, const uint8_t *value, const field &f) {
|
||||
iter.log_value(f.key);
|
||||
auto i = reinterpret_cast<uint64_t *>(reinterpret_cast<char *>(&tweets.back()) + f.offset);
|
||||
if (auto error = numberparsing::parse_unsigned(value).get(*i)) {
|
||||
// If number parsing failed, check if it's null before returning the error
|
||||
if (!atomparsing::is_valid_null_atom(value)) { iter.log_error("expected number or null"); return error; }
|
||||
i = 0;
|
||||
}
|
||||
return SUCCESS;
|
||||
}
|
||||
simdjson_really_inline error_code sax_tweet_reader_visitor::parse_unsigned(json_iterator &iter, const uint8_t *value, const field &f) {
|
||||
iter.log_value(f.key);
|
||||
auto i = reinterpret_cast<uint64_t *>(reinterpret_cast<char *>(&tweets.back()) + f.offset);
|
||||
return numberparsing::parse_unsigned(value).get(*i);
|
||||
}
|
||||
simdjson_really_inline error_code sax_tweet_reader_visitor::parse_string(json_iterator &iter, const uint8_t *value, const field &f) {
|
||||
iter.log_value(f.key);
|
||||
auto s = reinterpret_cast<std::string_view *>(reinterpret_cast<char *>(&tweets.back()) + f.offset);
|
||||
return stringparsing::parse_string_to_buffer(value, current_string_buf_loc, *s);
|
||||
}
|
||||
|
||||
sax_tweet_reader_visitor::field_lookup sax_tweet_reader_visitor::fields{};
|
||||
|
||||
simdjson_really_inline uint8_t sax_tweet_reader_visitor::field_lookup::hash(const char * key, uint32_t depth) {
|
||||
// These shift numbers were chosen specifically because this yields only 2 collisions between
|
||||
// keys in twitter.json, leaves 0 as a distinct value, and has 0 collisions between keys we
|
||||
// actually care about.
|
||||
return uint8_t((key[0] << 0) ^ (key[1] << 3) ^ (key[2] << 3) ^ (key[3] << 1) ^ depth);
|
||||
}
|
||||
simdjson_really_inline sax_tweet_reader_visitor::field sax_tweet_reader_visitor::field_lookup::get(const uint8_t * key, containers c) {
|
||||
auto index = hash((const char *)key, uint32_t(c));
|
||||
auto entry = entries[index];
|
||||
// TODO if any key is > SIMDJSON_PADDING, this will access inaccessible memory!
|
||||
if (c != entry.container || memcmp(key, entry.key, entry.len)) { return entries[0]; }
|
||||
return entry;
|
||||
}
|
||||
simdjson_really_inline void sax_tweet_reader_visitor::field_lookup::add(const char * key, size_t len, containers c, field_type type, size_t offset) {
|
||||
auto index = hash(key, uint32_t(c));
|
||||
if (index == 0) {
|
||||
fprintf(stderr, "%s (depth %d) hashes to zero, which is used as 'missing value'\n", key, int(c));
|
||||
assert(false);
|
||||
}
|
||||
if (entries[index].key) {
|
||||
fprintf(stderr, "%s (depth %d) collides with %s (depth %d) !\n", key, int(c), entries[index].key, int(entries[index].container));
|
||||
assert(false);
|
||||
}
|
||||
entries[index] = { key, len, offset, c, type };
|
||||
}
|
||||
simdjson_really_inline void sax_tweet_reader_visitor::field_lookup::neg(const char * const key, uint32_t depth) {
|
||||
auto index = hash(key, depth);
|
||||
if (entries[index].key) {
|
||||
fprintf(stderr, "%s (depth %d) conflicts with %s (depth %d) !\n", key, depth, entries[index].key, int(entries[index].container));
|
||||
}
|
||||
}
|
||||
|
||||
sax_tweet_reader_visitor::field_lookup::field_lookup() {
|
||||
add("\"statuses\"", std::strlen("\"statuses\""), containers::top_object, field_type::array, 0); // { "statuses": [...]
|
||||
#define TWEET_FIELD(KEY, TYPE) add("\"" #KEY "\"", std::strlen("\"" #KEY "\""), containers::tweet, TYPE, offsetof(tweet, KEY));
|
||||
TWEET_FIELD(id, field_type::unsigned_integer);
|
||||
TWEET_FIELD(in_reply_to_status_id, field_type::nullable_unsigned_integer);
|
||||
TWEET_FIELD(retweet_count, field_type::unsigned_integer);
|
||||
TWEET_FIELD(favorite_count, field_type::unsigned_integer);
|
||||
TWEET_FIELD(text, field_type::string);
|
||||
TWEET_FIELD(created_at, field_type::string);
|
||||
TWEET_FIELD(user, field_type::object)
|
||||
#undef TWEET_FIELD
|
||||
#define USER_FIELD(KEY, TYPE) add("\"" #KEY "\"", std::strlen("\"" #KEY "\""), containers::user, TYPE, offsetof(tweet, user)+offsetof(twitter_user, KEY));
|
||||
USER_FIELD(id, field_type::unsigned_integer);
|
||||
USER_FIELD(screen_name, field_type::string);
|
||||
#undef USER_FIELD
|
||||
|
||||
// Check for collisions with other (unused) hash keys in typical twitter JSON
|
||||
#define NEG(key, depth) neg("\"" #key "\"", depth);
|
||||
NEG(display_url, 9);
|
||||
NEG(expanded_url, 9);
|
||||
neg("\"h\":", 9);
|
||||
NEG(indices, 9);
|
||||
NEG(resize, 9);
|
||||
NEG(url, 9);
|
||||
neg("\"w\":", 9);
|
||||
NEG(display_url, 8);
|
||||
NEG(expanded_url, 8);
|
||||
neg("\"h\":", 8);
|
||||
NEG(indices, 8);
|
||||
NEG(large, 8);
|
||||
NEG(medium, 8);
|
||||
NEG(resize, 8);
|
||||
NEG(small, 8);
|
||||
NEG(thumb, 8);
|
||||
NEG(url, 8);
|
||||
neg("\"w\":", 8);
|
||||
NEG(display_url, 7);
|
||||
NEG(expanded_url, 7);
|
||||
NEG(id_str, 7);
|
||||
NEG(id, 7);
|
||||
NEG(indices, 7);
|
||||
NEG(large, 7);
|
||||
NEG(media_url_https, 7);
|
||||
NEG(media_url, 7);
|
||||
NEG(medium, 7);
|
||||
NEG(name, 7);
|
||||
NEG(sizes, 7);
|
||||
NEG(small, 7);
|
||||
NEG(source_status_id_str, 7);
|
||||
NEG(source_status_id, 7);
|
||||
NEG(thumb, 7);
|
||||
NEG(type, 7);
|
||||
NEG(url, 7);
|
||||
NEG(urls, 7);
|
||||
NEG(description, 6);
|
||||
NEG(display_url, 6);
|
||||
NEG(expanded_url, 6);
|
||||
NEG(id_str, 6);
|
||||
NEG(id, 6);
|
||||
NEG(indices, 6);
|
||||
NEG(media_url_https, 6);
|
||||
NEG(media_url, 6);
|
||||
NEG(name, 6);
|
||||
NEG(sizes, 6);
|
||||
NEG(source_status_id_str, 6);
|
||||
NEG(source_status_id, 6);
|
||||
NEG(type, 6);
|
||||
NEG(url, 6);
|
||||
NEG(urls, 6);
|
||||
NEG(contributors_enabled, 5);
|
||||
NEG(default_profile_image, 5);
|
||||
NEG(default_profile, 5);
|
||||
NEG(description, 5);
|
||||
NEG(entities, 5);
|
||||
NEG(favourites_count, 5);
|
||||
NEG(follow_request_sent, 5);
|
||||
NEG(followers_count, 5);
|
||||
NEG(following, 5);
|
||||
NEG(friends_count, 5);
|
||||
NEG(geo_enabled, 5);
|
||||
NEG(hashtags, 5);
|
||||
NEG(id_str, 5);
|
||||
NEG(id, 5);
|
||||
NEG(is_translation_enabled, 5);
|
||||
NEG(is_translator, 5);
|
||||
NEG(iso_language_code, 5);
|
||||
NEG(lang, 5);
|
||||
NEG(listed_count, 5);
|
||||
NEG(location, 5);
|
||||
NEG(media, 5);
|
||||
NEG(name, 5);
|
||||
NEG(notifications, 5);
|
||||
NEG(profile_background_color, 5);
|
||||
NEG(profile_background_image_url_https, 5);
|
||||
NEG(profile_background_image_url, 5);
|
||||
NEG(profile_background_tile, 5);
|
||||
NEG(profile_banner_url, 5);
|
||||
NEG(profile_image_url_https, 5);
|
||||
NEG(profile_image_url, 5);
|
||||
NEG(profile_link_color, 5);
|
||||
NEG(profile_sidebar_border_color, 5);
|
||||
NEG(profile_sidebar_fill_color, 5);
|
||||
NEG(profile_text_color, 5);
|
||||
NEG(profile_use_background_image, 5);
|
||||
NEG(protected, 5);
|
||||
NEG(result_type, 5);
|
||||
NEG(statuses_count, 5);
|
||||
NEG(symbols, 5);
|
||||
NEG(time_zone, 5);
|
||||
NEG(url, 5);
|
||||
NEG(urls, 5);
|
||||
NEG(user_mentions, 5);
|
||||
NEG(utc_offset, 5);
|
||||
NEG(verified, 5);
|
||||
NEG(contributors_enabled, 4);
|
||||
NEG(contributors, 4);
|
||||
NEG(coordinates, 4);
|
||||
NEG(default_profile_image, 4);
|
||||
NEG(default_profile, 4);
|
||||
NEG(description, 4);
|
||||
NEG(entities, 4);
|
||||
NEG(favorited, 4);
|
||||
NEG(favourites_count, 4);
|
||||
NEG(follow_request_sent, 4);
|
||||
NEG(followers_count, 4);
|
||||
NEG(following, 4);
|
||||
NEG(friends_count, 4);
|
||||
NEG(geo_enabled, 4);
|
||||
NEG(geo, 4);
|
||||
NEG(hashtags, 4);
|
||||
NEG(id_str, 4);
|
||||
NEG(in_reply_to_screen_name, 4);
|
||||
NEG(in_reply_to_status_id_str, 4);
|
||||
NEG(in_reply_to_user_id_str, 4);
|
||||
NEG(in_reply_to_user_id, 4);
|
||||
NEG(is_translation_enabled, 4);
|
||||
NEG(is_translator, 4);
|
||||
NEG(iso_language_code, 4);
|
||||
NEG(lang, 4);
|
||||
NEG(listed_count, 4);
|
||||
NEG(location, 4);
|
||||
NEG(media, 4);
|
||||
NEG(metadata, 4);
|
||||
NEG(name, 4);
|
||||
NEG(notifications, 4);
|
||||
NEG(place, 4);
|
||||
NEG(possibly_sensitive, 4);
|
||||
NEG(profile_background_color, 4);
|
||||
NEG(profile_background_image_url_https, 4);
|
||||
NEG(profile_background_image_url, 4);
|
||||
NEG(profile_background_tile, 4);
|
||||
NEG(profile_banner_url, 4);
|
||||
NEG(profile_image_url_https, 4);
|
||||
NEG(profile_image_url, 4);
|
||||
NEG(profile_link_color, 4);
|
||||
NEG(profile_sidebar_border_color, 4);
|
||||
NEG(profile_sidebar_fill_color, 4);
|
||||
NEG(profile_text_color, 4);
|
||||
NEG(profile_use_background_image, 4);
|
||||
NEG(protected, 4);
|
||||
NEG(result_type, 4);
|
||||
NEG(retweeted, 4);
|
||||
NEG(source, 4);
|
||||
NEG(statuses_count, 4);
|
||||
NEG(symbols, 4);
|
||||
NEG(time_zone, 4);
|
||||
NEG(truncated, 4);
|
||||
NEG(url, 4);
|
||||
NEG(urls, 4);
|
||||
NEG(user_mentions, 4);
|
||||
NEG(utc_offset, 4);
|
||||
NEG(verified, 4);
|
||||
NEG(contributors, 3);
|
||||
NEG(coordinates, 3);
|
||||
NEG(entities, 3);
|
||||
NEG(favorited, 3);
|
||||
NEG(geo, 3);
|
||||
NEG(id_str, 3);
|
||||
NEG(in_reply_to_screen_name, 3);
|
||||
NEG(in_reply_to_status_id_str, 3);
|
||||
NEG(in_reply_to_user_id_str, 3);
|
||||
NEG(in_reply_to_user_id, 3);
|
||||
NEG(lang, 3);
|
||||
NEG(metadata, 3);
|
||||
NEG(place, 3);
|
||||
NEG(possibly_sensitive, 3);
|
||||
NEG(retweeted_status, 3);
|
||||
NEG(retweeted, 3);
|
||||
NEG(source, 3);
|
||||
NEG(truncated, 3);
|
||||
NEG(completed_in, 2);
|
||||
NEG(count, 2);
|
||||
NEG(max_id_str, 2);
|
||||
NEG(max_id, 2);
|
||||
NEG(next_results, 2);
|
||||
NEG(query, 2);
|
||||
NEG(refresh_url, 2);
|
||||
NEG(since_id_str, 2);
|
||||
NEG(since_id, 2);
|
||||
NEG(search_metadata, 1);
|
||||
#undef NEG
|
||||
}
|
||||
|
||||
// sax_tweet_reader_visitor::field_lookup::find_min() {
|
||||
// int min_count = 100000;
|
||||
// for (int a=0;a<4;a++) {
|
||||
// for (int b=0;b<4;b++) {
|
||||
// for (int c=0;c<4;c++) {
|
||||
// sax_tweet_reader_visitor::field_lookup fields(a,b,c);
|
||||
// if (fields.collision_count) { continue; }
|
||||
// if (fields.zero_emission) { continue; }
|
||||
// if (fields.conflict_count < min_count) { printf("min=%d,%d,%d (%d)", a, b, c, fields.conflict_count); }
|
||||
// }
|
||||
// }
|
||||
// }
|
||||
// }
|
||||
|
||||
} // namespace partial_tweets
|
||||
@@ -0,0 +1,57 @@
|
||||
#pragma once
|
||||
|
||||
#include "simdjson.h"
|
||||
#include "twitter_user.h"
|
||||
|
||||
namespace partial_tweets {
|
||||
|
||||
// {
|
||||
// "statuses": [
|
||||
// {
|
||||
// "created_at": "Sun Aug 31 00:29:15 +0000 2014",
|
||||
// "id": 505874924095815700,
|
||||
// "text": "@aym0566x \n\n名前:前田あゆみ\n第一印象:なんか怖っ!\n今の印象:とりあえずキモい。噛み合わない\n好きなところ:ぶすでキモいとこ😋✨✨\n思い出:んーーー、ありすぎ😊❤️\nLINE交換できる?:あぁ……ごめん✋\nトプ画をみて:照れますがな😘✨\n一言:お前は一生もんのダチ💖",
|
||||
// "in_reply_to_status_id": null,
|
||||
// "user": {
|
||||
// "id": 1186275104,
|
||||
// "screen_name": "ayuu0123"
|
||||
// },
|
||||
// "retweet_count": 0,
|
||||
// "favorite_count": 0
|
||||
// }
|
||||
// ]
|
||||
// }
|
||||
|
||||
struct tweet {
|
||||
std::string_view created_at{};
|
||||
uint64_t id{};
|
||||
std::string_view text{};
|
||||
uint64_t in_reply_to_status_id{};
|
||||
twitter_user user{};
|
||||
uint64_t retweet_count{};
|
||||
uint64_t favorite_count{};
|
||||
simdjson_really_inline bool operator==(const tweet &other) const {
|
||||
return created_at == other.created_at &&
|
||||
id == other.id &&
|
||||
text == other.text &&
|
||||
in_reply_to_status_id == other.in_reply_to_status_id &&
|
||||
user == other.user &&
|
||||
retweet_count == other.retweet_count &&
|
||||
favorite_count == other.favorite_count;
|
||||
}
|
||||
simdjson_really_inline bool operator!=(const tweet &other) const { return !(*this == other); }
|
||||
};
|
||||
|
||||
simdjson_unused static std::ostream &operator<<(std::ostream &o, const tweet &t) {
|
||||
o << "created_at: " << t.created_at << std::endl;
|
||||
o << "id: " << t.id << std::endl;
|
||||
o << "text: " << t.text << std::endl;
|
||||
o << "in_reply_to_status_id: " << t.in_reply_to_status_id << std::endl;
|
||||
o << "user.id: " << t.user.id << std::endl;
|
||||
o << "user.screen_name: " << t.user.screen_name << std::endl;
|
||||
o << "retweet_count: " << t.retweet_count << std::endl;
|
||||
o << "favorite_count: " << t.favorite_count << std::endl;
|
||||
return o;
|
||||
}
|
||||
|
||||
} // namespace partial_tweets
|
||||
@@ -0,0 +1,16 @@
|
||||
#pragma once
|
||||
#include "simdjson.h"
|
||||
|
||||
namespace partial_tweets {
|
||||
|
||||
struct twitter_user {
|
||||
uint64_t id{};
|
||||
std::string_view screen_name{};
|
||||
|
||||
bool operator==(const twitter_user &other) const {
|
||||
return id == other.id &&
|
||||
screen_name == other.screen_name;
|
||||
}
|
||||
};
|
||||
|
||||
} // namespace partial_tweets
|
||||
@@ -0,0 +1,114 @@
|
||||
#include <cstdio>
|
||||
#include <iostream>
|
||||
#include <memory>
|
||||
#include <stdexcept>
|
||||
#include <string>
|
||||
#include <sstream>
|
||||
#include <array>
|
||||
#include <algorithm>
|
||||
#include <vector>
|
||||
#include <cmath>
|
||||
|
||||
#ifdef _WIN32
|
||||
#define popen _popen
|
||||
#define pclose _pclose
|
||||
#endif
|
||||
|
||||
int closepipe(FILE *pipe) {
|
||||
int exit_code = pclose(pipe);
|
||||
if (exit_code != EXIT_SUCCESS) {
|
||||
std::cerr << "Error " << exit_code << " running benchmark command!" << std::endl;
|
||||
exit(EXIT_FAILURE);
|
||||
};
|
||||
return exit_code;
|
||||
}
|
||||
|
||||
std::string exec(const char* cmd) {
|
||||
std::cerr << cmd << std::endl;
|
||||
std::array<char, 128> buffer;
|
||||
std::string result;
|
||||
std::unique_ptr<FILE, decltype(&closepipe)> pipe(popen(cmd, "r"), closepipe);
|
||||
if (!pipe) {
|
||||
std::cerr << "popen() failed!" << std::endl;
|
||||
abort();
|
||||
}
|
||||
while (fgets(buffer.data(), int(buffer.size()), pipe.get()) != nullptr) {
|
||||
result += buffer.data();
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
double readThroughput(std::string parseOutput) {
|
||||
std::istringstream output(parseOutput);
|
||||
std::string line;
|
||||
double result = 0;
|
||||
int numResults = 0;
|
||||
while (std::getline(output, line)) {
|
||||
std::string::size_type pos = 0;
|
||||
for (int i=0; i<5; i++) {
|
||||
pos = line.find('\t', pos);
|
||||
if (pos == std::string::npos) {
|
||||
std::cerr << "Command printed out a line with less than 5 fields in it:\n" << line << std::endl;
|
||||
}
|
||||
pos++;
|
||||
}
|
||||
result += std::stod(line.substr(pos));
|
||||
numResults++;
|
||||
}
|
||||
if (numResults == 0) {
|
||||
std::cerr << "No results returned from benchmark command!" << std::endl;
|
||||
exit(EXIT_FAILURE);
|
||||
}
|
||||
return result / numResults;
|
||||
}
|
||||
|
||||
const double INTERLEAVED_ATTEMPTS = 7;
|
||||
|
||||
int main(int argc, const char *argv[]) {
|
||||
if (argc < 3) {
|
||||
std::cerr << "Usage: " << argv[0] << " <old parse exe> <new parse exe> [<parse arguments>]" << std::endl;
|
||||
return 1;
|
||||
}
|
||||
|
||||
std::string newCommand = argv[1];
|
||||
std::string refCommand = argv[2];
|
||||
for (int i=3; i<argc; i++) {
|
||||
newCommand += " ";
|
||||
newCommand += argv[i];
|
||||
refCommand += " ";
|
||||
refCommand += argv[i];
|
||||
}
|
||||
|
||||
std::vector<double> ref;
|
||||
std::vector<double> newcode;
|
||||
for (int attempt=0; attempt < INTERLEAVED_ATTEMPTS; attempt++) {
|
||||
std::cout << "Attempt #" << (attempt+1) << " of " << INTERLEAVED_ATTEMPTS << std::endl;
|
||||
|
||||
// Read new throughput
|
||||
double newThroughput = readThroughput(exec(newCommand.c_str()));
|
||||
std::cout << "New throughput: " << newThroughput << std::endl;
|
||||
newcode.push_back(newThroughput);
|
||||
|
||||
// Read reference throughput
|
||||
double referenceThroughput = readThroughput(exec(refCommand.c_str()));
|
||||
std::cout << "Ref throughput: " << referenceThroughput << std::endl;
|
||||
ref.push_back(referenceThroughput);
|
||||
}
|
||||
// we check if the maximum of newcode is lower than minimum of ref, if so we have a problem so fail!
|
||||
double worseref = *std::min_element(ref.begin(), ref.end());
|
||||
double bestnewcode = *std::max_element(newcode.begin(), newcode.end());
|
||||
double bestref = *std::max_element(ref.begin(), ref.end());
|
||||
double worsenewcode = *std::min_element(newcode.begin(), newcode.end());
|
||||
std::cout << "The new code has a throughput in " << worsenewcode << " -- " << bestnewcode << std::endl;
|
||||
std::cout << "The reference code has a throughput in " << worseref << " -- " << bestref << std::endl;
|
||||
if(bestnewcode < worseref) {
|
||||
std::cerr << "You probably have a performance degradation." << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
if(bestnewcode < worseref) {
|
||||
std::cout << "You probably have a performance gain." << std::endl;
|
||||
return EXIT_SUCCESS;
|
||||
}
|
||||
std::cout << "There is no obvious performance difference. A manual check might be needed." << std::endl;
|
||||
return EXIT_SUCCESS;
|
||||
}
|
||||
+101
-100
@@ -1,15 +1,10 @@
|
||||
#include <iostream>
|
||||
#ifndef _MSC_VER
|
||||
#include <unistd.h>
|
||||
#endif
|
||||
#include "simdjson/jsonioutil.h"
|
||||
#include "simdjson/jsonparser.h"
|
||||
#include "simdjson.h"
|
||||
#ifdef __linux__
|
||||
#include "linux-perf-events.h"
|
||||
#endif
|
||||
|
||||
using namespace std;
|
||||
|
||||
size_t count_nonasciibytes(const uint8_t *input, size_t length) {
|
||||
size_t count = 0;
|
||||
for (size_t i = 0; i < length; i++) {
|
||||
@@ -31,7 +26,7 @@ struct stat_s {
|
||||
size_t float_count;
|
||||
size_t string_count;
|
||||
size_t backslash_count;
|
||||
size_t nonasciibyte_count;
|
||||
size_t non_ascii_byte_count;
|
||||
size_t object_count;
|
||||
size_t array_count;
|
||||
size_t null_count;
|
||||
@@ -44,90 +39,96 @@ struct stat_s {
|
||||
|
||||
using stat_t = struct stat_s;
|
||||
|
||||
stat_t simdjson_computestats(const std::string_view &p) {
|
||||
stat_t answer;
|
||||
ParsedJson pj = build_parsed_json(p);
|
||||
answer.valid = pj.isValid();
|
||||
if (!answer.valid) {
|
||||
|
||||
|
||||
simdjson_really_inline void simdjson_process_atom(stat_t &s,
|
||||
simdjson::dom::element element) {
|
||||
if (element.is<int64_t>()) {
|
||||
s.integer_count++;
|
||||
} else if(element.is<std::string_view>()) {
|
||||
s.string_count++;
|
||||
} else if(element.is<double>()) {
|
||||
s.float_count++;
|
||||
} else if (element.is<bool>()) {
|
||||
simdjson::error_code err;
|
||||
bool v;
|
||||
err = element.get(v);
|
||||
if (v) {
|
||||
s.true_count++;
|
||||
} else {
|
||||
s.false_count++;
|
||||
}
|
||||
} else if (element.is_null()) {
|
||||
s.null_count++;
|
||||
}
|
||||
}
|
||||
|
||||
void simdjson_recurse(stat_t &s, simdjson::dom::element element) {
|
||||
simdjson::error_code error;
|
||||
if (element.is<simdjson::dom::array>()) {
|
||||
s.array_count++;
|
||||
simdjson::dom::array array;
|
||||
if ((error = element.get(array))) { std::cerr << error << std::endl; abort(); }
|
||||
for (auto child : array) {
|
||||
if (child.is<simdjson::dom::array>() || child.is<simdjson::dom::object>()) {
|
||||
simdjson_recurse(s, child);
|
||||
} else {
|
||||
simdjson_process_atom(s, child);
|
||||
}
|
||||
}
|
||||
} else if (element.is<simdjson::dom::object>()) {
|
||||
s.object_count++;
|
||||
simdjson::dom::object object;
|
||||
if ((error = element.get(object))) { std::cerr << error << std::endl; abort(); }
|
||||
for (auto field : object) {
|
||||
s.string_count++; // for key
|
||||
if (field.value.is<simdjson::dom::array>() || field.value.is<simdjson::dom::object>()) {
|
||||
simdjson_recurse(s, field.value);
|
||||
} else {
|
||||
simdjson_process_atom(s, field.value);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
simdjson_process_atom(s, element);
|
||||
}
|
||||
}
|
||||
|
||||
stat_t simdjson_compute_stats(const simdjson::padded_string &p) {
|
||||
stat_t answer{};
|
||||
simdjson::dom::parser parser;
|
||||
simdjson::dom::element doc;
|
||||
auto error = parser.parse(p).get(doc);
|
||||
if (error) {
|
||||
answer.valid = false;
|
||||
return answer;
|
||||
}
|
||||
answer.backslash_count = count_backslash(reinterpret_cast<const uint8_t *>(p.data()), p.size());
|
||||
answer.nonasciibyte_count =
|
||||
count_nonasciibytes(reinterpret_cast<const uint8_t *>(p.data()), p.size());
|
||||
answer.valid = true;
|
||||
answer.backslash_count =
|
||||
count_backslash(reinterpret_cast<const uint8_t *>(p.data()), p.size());
|
||||
answer.non_ascii_byte_count = count_nonasciibytes(
|
||||
reinterpret_cast<const uint8_t *>(p.data()), p.size());
|
||||
answer.byte_count = p.size();
|
||||
answer.integer_count = 0;
|
||||
answer.float_count = 0;
|
||||
answer.object_count = 0;
|
||||
answer.array_count = 0;
|
||||
answer.null_count = 0;
|
||||
answer.true_count = 0;
|
||||
answer.false_count = 0;
|
||||
answer.string_count = 0;
|
||||
answer.structural_indexes_count = pj.n_structural_indexes;
|
||||
size_t tapeidx = 0;
|
||||
uint64_t tape_val = pj.tape[tapeidx++];
|
||||
uint8_t type = (tape_val >> 56);
|
||||
size_t howmany = 0;
|
||||
assert(type == 'r');
|
||||
howmany = tape_val & JSONVALUEMASK;
|
||||
for (; tapeidx < howmany; tapeidx++) {
|
||||
tape_val = pj.tape[tapeidx];
|
||||
// uint64_t payload = tape_val & JSONVALUEMASK;
|
||||
type = (tape_val >> 56);
|
||||
switch (type) {
|
||||
case 'l': // we have a long int
|
||||
answer.integer_count++;
|
||||
tapeidx++; // skipping the integer
|
||||
break;
|
||||
case 'd': // we have a double
|
||||
answer.float_count++;
|
||||
tapeidx++; // skipping the double
|
||||
break;
|
||||
case 'n': // we have a null
|
||||
answer.null_count++;
|
||||
break;
|
||||
case 't': // we have a true
|
||||
answer.true_count++;
|
||||
break;
|
||||
case 'f': // we have a false
|
||||
answer.false_count++;
|
||||
break;
|
||||
case '{': // we have an object
|
||||
answer.object_count++;
|
||||
break;
|
||||
case '}': // we end an object
|
||||
break;
|
||||
case '[': // we start an array
|
||||
answer.array_count++;
|
||||
break;
|
||||
case ']': // we end an array
|
||||
break;
|
||||
case '"': // we have a string
|
||||
answer.string_count++;
|
||||
break;
|
||||
default:
|
||||
break; // ignore
|
||||
}
|
||||
}
|
||||
answer.structural_indexes_count = parser.implementation->n_structural_indexes;
|
||||
simdjson_recurse(answer, doc);
|
||||
return answer;
|
||||
}
|
||||
|
||||
int main(int argc, char *argv[]) {
|
||||
#ifndef _MSC_VER
|
||||
int c;
|
||||
while ((c = getopt(argc, argv, "")) != -1) {
|
||||
int c;
|
||||
while ((c = getopt(argc, argv, "")) != -1) {
|
||||
switch (c) {
|
||||
|
||||
default:
|
||||
abort();
|
||||
}
|
||||
}
|
||||
}
|
||||
#else
|
||||
int optind = 1;
|
||||
#endif
|
||||
if (optind >= argc) {
|
||||
cerr << "Reads json, prints stats. " << endl;
|
||||
cerr << "Usage: " << argv[0] << " <jsonfile>" << endl;
|
||||
std::cerr << "Reads json, prints stats. " << std::endl;
|
||||
std::cerr << "Usage: " << argv[0] << " <jsonfile>" << std::endl;
|
||||
|
||||
exit(1);
|
||||
}
|
||||
@@ -136,70 +137,70 @@ int main(int argc, char *argv[]) {
|
||||
std::cerr << "warning: ignoring everything after " << argv[optind + 1]
|
||||
<< std::endl;
|
||||
}
|
||||
std::string_view p;
|
||||
try {
|
||||
p = get_corpus(filename);
|
||||
} catch (const std::exception &e) { // caught by reference to base
|
||||
simdjson::padded_string p;
|
||||
auto error = simdjson::padded_string::load(filename).get(p);
|
||||
if (error) {
|
||||
std::cerr << "Could not load the file " << filename << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
stat_t s = simdjson_computestats(p);
|
||||
stat_t s = simdjson_compute_stats(p);
|
||||
if (!s.valid) {
|
||||
std::cerr << "not a valid JSON" << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
|
||||
printf("# integer_count float_count string_count backslash_count "
|
||||
"nonasciibyte_count object_count array_count null_count true_count "
|
||||
"non_ascii_byte_count object_count array_count null_count true_count "
|
||||
"false_count byte_count structural_indexes_count ");
|
||||
#ifdef __linux__
|
||||
printf(
|
||||
" stage1_cycle_count stage1_instruction_count stage2_cycle_count "
|
||||
" stage2_instruction_count stage3_cycle_count stage3_instruction_count ");
|
||||
printf(" stage1_cycle_count stage1_instruction_count stage2_cycle_count "
|
||||
" stage2_instruction_count stage3_cycle_count "
|
||||
"stage3_instruction_count ");
|
||||
#else
|
||||
printf("(you are not under linux, so perf counters are disaabled)");
|
||||
#endif
|
||||
printf("\n");
|
||||
printf("%zu %zu %zu %zu %zu %zu %zu %zu %zu %zu %zu %zu ", s.integer_count,
|
||||
s.float_count, s.string_count, s.backslash_count, s.nonasciibyte_count,
|
||||
s.object_count, s.array_count, s.null_count, s.true_count,
|
||||
s.false_count, s.byte_count, s.structural_indexes_count);
|
||||
s.float_count, s.string_count, s.backslash_count,
|
||||
s.non_ascii_byte_count, s.object_count, s.array_count, s.null_count,
|
||||
s.true_count, s.false_count, s.byte_count, s.structural_indexes_count);
|
||||
#ifdef __linux__
|
||||
ParsedJson pj;
|
||||
bool allocok = pj.allocateCapacity(p.size());
|
||||
if (!allocok) {
|
||||
std::cerr << "failed to allocate memory" << std::endl;
|
||||
simdjson::dom::parser parser;
|
||||
simdjson::error_code alloc_error = parser.allocate(p.size());
|
||||
if (alloc_error) {
|
||||
std::cerr << alloc_error << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
const uint32_t iterations = p.size() < 1 * 1000 * 1000 ? 1000 : 50;
|
||||
vector<int> evts;
|
||||
std::vector<int> evts;
|
||||
evts.push_back(PERF_COUNT_HW_CPU_CYCLES);
|
||||
evts.push_back(PERF_COUNT_HW_INSTRUCTIONS);
|
||||
LinuxEvents<PERF_TYPE_HARDWARE> unified(evts);
|
||||
unsigned long cy1 = 0, cy2 = 0;
|
||||
unsigned long cl1 = 0, cl2 = 0;
|
||||
vector<unsigned long long> results;
|
||||
std::vector<unsigned long long> results;
|
||||
results.resize(evts.size());
|
||||
for (uint32_t i = 0; i < iterations; i++) {
|
||||
unified.start();
|
||||
bool isok = find_structural_bits(p.data(), p.size(), pj);
|
||||
// The default template is simdjson::architecture::NATIVE.
|
||||
bool isok = (parser.implementation->stage1((const uint8_t *)p.data(), p.size(), false) == simdjson::SUCCESS);
|
||||
unified.end(results);
|
||||
|
||||
|
||||
cy1 += results[0];
|
||||
cl1 += results[1];
|
||||
|
||||
|
||||
unified.start();
|
||||
isok = isok && unified_machine(p.data(), p.size(), pj);
|
||||
isok = isok && (parser.implementation->stage2(parser.doc) == simdjson::SUCCESS);
|
||||
unified.end(results);
|
||||
|
||||
|
||||
cy2 += results[0];
|
||||
cl2 += results[1];
|
||||
if(!isok) {
|
||||
if (!isok) {
|
||||
std::cerr << "failure?" << std::endl;
|
||||
}
|
||||
}
|
||||
printf("%f %f %f %f ", cy1 * 1.0 / iterations, cl1 * 1.0 / iterations,
|
||||
cy2 * 1.0 / iterations, cl2 * 1.0 / iterations);
|
||||
printf("%f %f %f %f ", static_cast<double>(cy1) / static_cast<double>(iterations), static_cast<double>(cl1) / static_cast<double>(iterations),
|
||||
static_cast<double>(cy2) / static_cast<double>(iterations), static_cast<double>(cl2) / static_cast<double>(iterations));
|
||||
#endif // __linux__
|
||||
printf("\n");
|
||||
return EXIT_SUCCESS;
|
||||
|
||||
@@ -0,0 +1,49 @@
|
||||
# Helper so we don't have to repeat ourselves so much
|
||||
# Usage: add_cpp_test(testname [COMPILE_ONLY] [SOURCES a.cpp b.cpp ...] [LABELS acceptance per_implementation ...])
|
||||
# SOURCES defaults to testname.cpp if not specified.
|
||||
function(add_cpp_test TEST_NAME)
|
||||
# Parse arguments
|
||||
cmake_parse_arguments(PARSE_ARGV 1 ARGS "COMPILE_ONLY;LIBRARY;WILL_FAIL" "" "SOURCES;LABELS")
|
||||
if (NOT ARGS_SOURCES)
|
||||
list(APPEND ARGS_SOURCES ${TEST_NAME}.cpp)
|
||||
endif()
|
||||
if (ARGS_COMPILE_ONLY)
|
||||
list(APPEND ${ARGS_LABELS} compile)
|
||||
endif()
|
||||
|
||||
# Add the compile target
|
||||
if (ARGS_LIBRARY)
|
||||
add_library(${TEST_NAME} STATIC ${ARGS_SOURCES})
|
||||
else(ARGS_LIBRARY)
|
||||
add_executable(${TEST_NAME} ${ARGS_SOURCES})
|
||||
endif(ARGS_LIBRARY)
|
||||
|
||||
# Add test
|
||||
if (ARGS_COMPILE_ONLY OR ARGS_LIBRARY)
|
||||
add_test(
|
||||
NAME ${TEST_NAME}
|
||||
COMMAND ${CMAKE_COMMAND} --build . --target ${TEST_NAME} --config $<CONFIGURATION>
|
||||
WORKING_DIRECTORY ${PROJECT_BINARY_DIR}
|
||||
)
|
||||
set_target_properties(${TEST_NAME} PROPERTIES EXCLUDE_FROM_ALL TRUE EXCLUDE_FROM_DEFAULT_BUILD TRUE)
|
||||
else()
|
||||
add_test(${TEST_NAME} ${TEST_NAME})
|
||||
endif()
|
||||
|
||||
if (ARGS_LABELS)
|
||||
set_property(TEST ${TEST_NAME} APPEND PROPERTY LABELS ${ARGS_LABELS})
|
||||
endif()
|
||||
|
||||
if (ARGS_WILL_FAIL)
|
||||
set_property(TEST ${TEST_NAME} PROPERTY WILL_FAIL TRUE)
|
||||
endif()
|
||||
endfunction()
|
||||
|
||||
function(add_compile_only_test TEST_NAME)
|
||||
add_test(
|
||||
NAME ${TEST_NAME}
|
||||
COMMAND ${CMAKE_COMMAND} --build . --target ${TEST_NAME} --config $<CONFIGURATION>
|
||||
WORKING_DIRECTORY ${PROJECT_BINARY_DIR}
|
||||
)
|
||||
set_target_properties(${TEST_NAME} PROPERTIES EXCLUDE_FROM_ALL TRUE EXCLUDE_FROM_DEFAULT_BUILD TRUE)
|
||||
endfunction()
|
||||
@@ -0,0 +1,236 @@
|
||||
|
||||
if(CMAKE_SOURCE_DIR STREQUAL CMAKE_CURRENT_SOURCE_DIR)
|
||||
message (STATUS "The simdjson repository appears to be standalone.")
|
||||
option(SIMDJSON_JUST_LIBRARY "Build just the library, omit tests, tools and benchmarks" OFF)
|
||||
message (STATUS "By default, we attempt to build everything.")
|
||||
else()
|
||||
message (STATUS "The simdjson repository appears to be used as a subdirectory.")
|
||||
option(SIMDJSON_JUST_LIBRARY "Build just the library, omit tests, tools and benchmarks" ON)
|
||||
message (STATUS "By default, we just build the library.")
|
||||
endif()
|
||||
|
||||
#
|
||||
# Flags used by exes and by the simdjson library (project-wide flags)
|
||||
#
|
||||
add_library(simdjson-flags INTERFACE)
|
||||
add_library(simdjson-internal-flags INTERFACE)
|
||||
target_link_libraries(simdjson-internal-flags INTERFACE simdjson-flags)
|
||||
|
||||
option(SIMDJSON_SANITIZE "Sanitize addresses" OFF)
|
||||
if(SIMDJSON_SANITIZE)
|
||||
target_compile_options(simdjson-flags INTERFACE -fsanitize=address -fno-omit-frame-pointer -fsanitize=undefined -fno-sanitize-recover=all)
|
||||
target_link_libraries(simdjson-flags INTERFACE -fsanitize=address -fno-omit-frame-pointer -fsanitize=undefined -fno-sanitize-recover=all)
|
||||
|
||||
# Ubuntu bug for GCC 5.0+ (safe for all versions)
|
||||
if (CMAKE_COMPILER_IS_GNUCC)
|
||||
target_link_libraries(simdjson-flags INTERFACE -fuse-ld=gold)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if(SIMDJSON_SANITIZE_THREADS)
|
||||
target_compile_options(simdjson-flags INTERFACE -fsanitize=thread -fsanitize=undefined -fno-sanitize-recover=all)
|
||||
target_link_libraries(simdjson-flags INTERFACE -fsanitize=thread -fsanitize=undefined -fno-sanitize-recover=all)
|
||||
|
||||
# Ubuntu bug for GCC 5.0+ (safe for all versions)
|
||||
if (CMAKE_COMPILER_IS_GNUCC)
|
||||
target_link_libraries(simdjson-flags INTERFACE -fuse-ld=gold)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if (NOT CMAKE_BUILD_TYPE)
|
||||
message(STATUS "No build type selected, default to Release")
|
||||
set(CMAKE_BUILD_TYPE Release CACHE STRING "Choose the type of build." FORCE)
|
||||
if(SIMDJSON_SANITIZE)
|
||||
message(WARNING "No build type selected and you have enabled the sanitizer. Consider setting CMAKE_BUILD_TYPE to Debug to help identify the eventual problems.")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if(MSVC)
|
||||
option(SIMDJSON_BUILD_STATIC "Build a static library" ON) # turning it on disables the production of a dynamic library
|
||||
else()
|
||||
option(SIMDJSON_BUILD_STATIC "Build a static library" OFF) # turning it on disables the production of a dynamic library
|
||||
option(SIMDJSON_USE_LIBCPP "Use the libc++ library" OFF)
|
||||
endif()
|
||||
|
||||
set(CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/tools/cmake")
|
||||
|
||||
# We compile tools, tests, etc. with C++ 17. Override yourself if you need on a target.
|
||||
set(CMAKE_CXX_STANDARD 17)
|
||||
set(CMAKE_CXX_STANDARD_REQUIRED ON)
|
||||
set(CMAKE_CXX_EXTENSIONS OFF)
|
||||
set(CMAKE_MACOSX_RPATH OFF)
|
||||
set(CMAKE_THREAD_PREFER_PTHREAD ON)
|
||||
set(THREADS_PREFER_PTHREAD_FLAG ON)
|
||||
|
||||
# LTO seems to create all sorts of fun problems. Let us
|
||||
# disable temporarily.
|
||||
#include(CheckIPOSupported)
|
||||
#check_ipo_supported(RESULT ltoresult)
|
||||
#if(ltoresult)
|
||||
# set(CMAKE_INTERPROCEDURAL_OPTIMIZATION TRUE)
|
||||
#endif()
|
||||
|
||||
option(SIMDJSON_VISUAL_STUDIO_BUILD_WITH_DEBUG_INFO_FOR_PROFILING "Under Visual Studio, add Zi to the compile flag and DEBUG to the link file to add debugging information to the release build for easier profiling inside tools like VTune" OFF)
|
||||
if(MSVC)
|
||||
if("${MSVC_TOOLSET_VERSION}" STREQUAL "140")
|
||||
# Visual Studio 2015 issues warnings and we tolerate it, cmake -G"Visual Studio 14" ..
|
||||
target_compile_options(simdjson-internal-flags INTERFACE /W0 /sdl)
|
||||
else()
|
||||
# Recent version of Visual Studio expected (2017, 2019...). Prior versions are unsupported.
|
||||
target_compile_options(simdjson-internal-flags INTERFACE /WX /W3 /sdl /w34714) # https://docs.microsoft.com/en-us/cpp/error-messages/compiler-warnings/compiler-warning-level-4-c4714?view=vs-2019
|
||||
endif()
|
||||
if(SIMDJSON_VISUAL_STUDIO_BUILD_WITH_DEBUG_INFO_FOR_PROFILING)
|
||||
target_link_options(simdjson-flags INTERFACE /DEBUG )
|
||||
target_compile_options(simdjson-flags INTERFACE /Zi)
|
||||
endif()
|
||||
else()
|
||||
if(NOT WIN32)
|
||||
target_compile_options(simdjson-internal-flags INTERFACE -fPIC)
|
||||
endif()
|
||||
target_compile_options(simdjson-internal-flags INTERFACE -Werror -Wall -Wextra -Weffc++)
|
||||
target_compile_options(simdjson-internal-flags INTERFACE -Wsign-compare -Wshadow -Wwrite-strings -Wpointer-arith -Winit-self -Wconversion -Wno-sign-conversion)
|
||||
endif()
|
||||
|
||||
#
|
||||
# Optional flags
|
||||
#
|
||||
|
||||
#
|
||||
# Implementation selection
|
||||
#
|
||||
set(SIMDJSON_ALL_IMPLEMENTATIONS "fallback;westmere;haswell;arm64;ppc64")
|
||||
|
||||
set(SIMDJSON_IMPLEMENTATION "" CACHE STRING "Semicolon-separated list of implementations to include (${SIMDJSON_ALL_IMPLEMENTATIONS}). If this is not set, any implementations that are supported at compile time and may be selected at runtime will be included.")
|
||||
foreach(implementation ${SIMDJSON_IMPLEMENTATION})
|
||||
if(NOT (implementation IN_LIST SIMDJSON_ALL_IMPLEMENTATIONS))
|
||||
message(ERROR "Implementation ${implementation} not supported by simdjson. Possible implementations: ${SIMDJSON_ALL_IMPLEMENTATIONS}")
|
||||
endif()
|
||||
endforeach(implementation)
|
||||
|
||||
set(SIMDJSON_EXCLUDE_IMPLEMENTATION "" CACHE STRING "Semicolon-separated list of implementations to exclude (haswell/westmere/arm64/ppc64/fallback). By default, excludes any implementations that are unsupported at compile time or cannot be selected at runtime.")
|
||||
foreach(implementation ${SIMDJSON_EXCLUDE_IMPLEMENTATION})
|
||||
if(NOT (implementation IN_LIST SIMDJSON_ALL_IMPLEMENTATIONS))
|
||||
message(ERROR "Implementation ${implementation} not supported by simdjson. Possible implementations: ${SIMDJSON_ALL_IMPLEMENTATIONS}")
|
||||
endif()
|
||||
endforeach(implementation)
|
||||
|
||||
foreach(implementation ${SIMDJSON_ALL_IMPLEMENTATIONS})
|
||||
string(TOUPPER ${implementation} implementation_upper)
|
||||
if(implementation IN_LIST SIMDJSON_EXCLUDE_IMPLEMENTATION)
|
||||
message(STATUS "Excluding implementation ${implementation} due to SIMDJSON_EXCLUDE_IMPLEMENTATION=${SIMDJSON_EXCLUDE_IMPLEMENTATION}")
|
||||
target_compile_definitions(simdjson-flags INTERFACE "SIMDJSON_IMPLEMENTATION_${implementation_upper}=0")
|
||||
elseif(implementation IN_LIST SIMDJSON_IMPLEMENTATION)
|
||||
message(STATUS "Including implementation ${implementation} due to SIMDJSON_IMPLEMENTATION=${SIMDJSON_IMPLEMENTATION}")
|
||||
target_compile_definitions(simdjson-flags INTERFACE "SIMDJSON_IMPLEMENTATION_${implementation_upper}=1")
|
||||
elseif(SIMDJSON_IMPLEMENTATION)
|
||||
message(STATUS "Excluding implementation ${implementation} due to SIMDJSON_IMPLEMENTATION=${SIMDJSON_IMPLEMENTATION}")
|
||||
target_compile_definitions(simdjson-flags INTERFACE "SIMDJSON_IMPLEMENTATION_${implementation_upper}=0")
|
||||
endif()
|
||||
endforeach(implementation)
|
||||
|
||||
# TODO make it so this generates the necessary compiler flags to select the given implementation as the builtin automatically!
|
||||
option(SIMDJSON_BUILTIN_IMPLEMENTATION "Select the implementation that will be used for user code. Defaults to the most universal implementation in SIMDJSON_IMPLEMENTATION (in the order ${SIMDJSON_ALL_IMPLEMENTATIONS}) if specified; otherwise, by default the compiler will pick the best implementation that can always be selected given the compiler flags." "")
|
||||
if(SIMDJSON_BUILTIN_IMPLEMENTATION)
|
||||
target_compile_definitions(simdjson-flags INTERFACE "SIMDJSON_BUILTIN_IMPLEMENTATION=${SIMDJSON_BUILTIN_IMPLEMENTATION}")
|
||||
else()
|
||||
# Pick the most universal implementation out of the selected implementations (if any)
|
||||
foreach(implementation ${SIMDJSON_ALL_IMPLEMENTATIONS})
|
||||
if(implementation IN_LIST SIMDJSON_IMPLEMENTATION AND NOT (implementation IN_LIST SIMDJSON_EXCLUDE_IMPLEMENTATION))
|
||||
message(STATUS "Selected implementation ${implementation} as builtin implementation based on ${SIMDJSON_IMPLEMENTATION}.")
|
||||
target_compile_definitions(simdjson-flags INTERFACE "SIMDJSON_BUILTIN_IMPLEMENTATION=${implementation}")
|
||||
break()
|
||||
endif()
|
||||
endforeach(implementation)
|
||||
endif(SIMDJSON_BUILTIN_IMPLEMENTATION)
|
||||
|
||||
option(SIMDJSON_IMPLEMENTATION_HASWELL "Include the haswell implementation" ON)
|
||||
if(NOT SIMDJSON_IMPLEMENTATION_HASWELL)
|
||||
message(DEPRECATION "SIMDJSON_IMPLEMENTATION_HASWELL is deprecated. Use SIMDJSON_IMPLEMENTATION=-haswell instead.")
|
||||
target_compile_definitions(simdjson-flags INTERFACE SIMDJSON_IMPLEMENTATION_HASWELL=0)
|
||||
endif()
|
||||
option(SIMDJSON_IMPLEMENTATION_WESTMERE "Include the westmere implementation" ON)
|
||||
if(NOT SIMDJSON_IMPLEMENTATION_WESTMERE)
|
||||
message(DEPRECATION "SIMDJSON_IMPLEMENTATION_WESTMERE is deprecated. SIMDJSON_IMPLEMENTATION=-westmere instead.")
|
||||
target_compile_definitions(simdjson-flags INTERFACE SIMDJSON_IMPLEMENTATION_WESTMERE=0)
|
||||
endif()
|
||||
option(SIMDJSON_IMPLEMENTATION_ARM64 "Include the arm64 implementation" ON)
|
||||
if(NOT SIMDJSON_IMPLEMENTATION_ARM64)
|
||||
message(DEPRECATION "SIMDJSON_IMPLEMENTATION_ARM64 is deprecated. Use SIMDJSON_IMPLEMENTATION=-arm64 instead.")
|
||||
target_compile_definitions(simdjson-flags INTERFACE SIMDJSON_IMPLEMENTATION_ARM64=0)
|
||||
endif()
|
||||
option(SIMDJSON_IMPLEMENTATION_PPC64 "Include the arm64 implementation" ON)
|
||||
if(NOT SIMDJSON_IMPLEMENTATION_PPC64)
|
||||
message(DEPRECATION "SIMDJSON_IMPLEMENTATION_PPC64 is deprecated. Use SIMDJSON_IMPLEMENTATION=-ppc64 instead.")
|
||||
target_compile_definitions(simdjson-flags INTERFACE SIMDJSON_IMPLEMENTATION_PPC64=0)
|
||||
endif()
|
||||
option(SIMDJSON_IMPLEMENTATION_FALLBACK "Include the fallback implementation" ON)
|
||||
if(NOT SIMDJSON_IMPLEMENTATION_FALLBACK)
|
||||
message(DEPRECATION "SIMDJSON_IMPLEMENTATION_FALLBACK is deprecated. Use SIMDJSON_IMPLEMENTATION=-fallback instead.")
|
||||
target_compile_definitions(simdjson-flags INTERFACE SIMDJSON_IMPLEMENTATION_FALLBACK=0)
|
||||
endif()
|
||||
|
||||
#
|
||||
# Other optional flags
|
||||
#
|
||||
option(SIMDJSON_ONDEMAND_SAFETY_RAILS "Validate ondemand user code at runtime to ensure it is being used correctly. Defaults to ON for debug builds, OFF for release builds." $<IF:$<CONFIG:DEBUG>,ON,OFF>)
|
||||
if(SIMDJSON_ONDEMAND_SAFETY_RAILS)
|
||||
message(STATUS "Ondemand safety rails enabled. Ondemand user code will be checked at runtime. This will be slower than normal!")
|
||||
target_compile_definitions(simdjson-flags INTERFACE SIMDJSON_ONDEMAND_SAFETY_RAILS)
|
||||
endif(SIMDJSON_ONDEMAND_SAFETY_RAILS)
|
||||
|
||||
option(SIMDJSON_BASH "Allow usage of bash within CMake" ON)
|
||||
|
||||
option(SIMDJSON_EXCEPTIONS "Enable simdjson's exception-throwing interface" ON)
|
||||
if(NOT SIMDJSON_EXCEPTIONS)
|
||||
message(STATUS "simdjson exception interface turned off. Code that does not check error codes will not compile.")
|
||||
target_compile_definitions(simdjson-flags INTERFACE SIMDJSON_EXCEPTIONS=0)
|
||||
if(MSVC)
|
||||
# CMake currently /EHsc as a default flag in CMAKE_CXX_FLAGS on MSVC. Replacing this with a more general abstraction is a WIP (see https://gitlab.kitware.com/cmake/cmake/-/issues/20610)
|
||||
# /EHs enables standard C++ stack unwinding when catching exceptions (non-structured exception handling)
|
||||
# /EHc used in conjection with /EHs indicates that extern "C" functions never throw (terminate-on-throw)
|
||||
# Here, we disable both with the - argument negation operator
|
||||
string(REPLACE "/EHsc" "/EHs-c-" CMAKE_CXX_FLAGS ${CMAKE_CXX_FLAGS})
|
||||
|
||||
# Because we cannot change the flag above on an invidual target (yet), the definition below must similarly be added globally
|
||||
add_definitions(-D_HAS_EXCEPTIONS=0)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
option(SIMDJSON_ENABLE_THREADS "Link with thread support" ON)
|
||||
if(SIMDJSON_ENABLE_THREADS)
|
||||
set(CMAKE_THREAD_PREFER_PTHREAD TRUE)
|
||||
set(THREADS_PREFER_PTHREAD_FLAG TRUE)
|
||||
find_package(Threads REQUIRED)
|
||||
target_link_libraries(simdjson-flags INTERFACE Threads::Threads)
|
||||
target_link_libraries(simdjson-flags INTERFACE ${CMAKE_THREAD_LIBS_INIT})
|
||||
target_compile_options(simdjson-flags INTERFACE ${CMAKE_THREAD_LIBS_INIT})
|
||||
target_compile_definitions(simdjson-flags INTERFACE SIMDJSON_THREADS_ENABLED=1) # This will be set in the code automatically.
|
||||
endif()
|
||||
|
||||
option(SIMDJSON_VERBOSE_LOGGING, "Enable verbose logging for internal simdjson library development." OFF)
|
||||
if (SIMDJSON_VERBOSE_LOGGING)
|
||||
target_compile_definitions(simdjson-flags INTERFACE SIMDJSON_VERBOSE_LOGGING=1)
|
||||
endif()
|
||||
|
||||
option(SIMDJSON_DISABLE_DEPRECATED_API "Disables deprecated APIs" Off)
|
||||
if (SIMDJSON_DISABLE_DEPRECATED_API)
|
||||
target_compile_definitions(simdjson-flags INTERFACE SIMDJSON_DISABLE_DEPRECATED_API=1)
|
||||
endif()
|
||||
|
||||
if(SIMDJSON_USE_LIBCPP)
|
||||
target_link_libraries(simdjson-flags INTERFACE -stdlib=libc++ -lc++abi)
|
||||
# instead of the above line, we could have used
|
||||
# set(CMAKE_EXE_LINKER_FLAGS "${CMAKE_EXE_LINKER_FLAGS} -stdlib=libc++ -lc++abi")
|
||||
# The next line is needed empirically.
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -stdlib=libc++")
|
||||
# we update CMAKE_SHARED_LINKER_FLAGS, this gets updated later as well
|
||||
set(CMAKE_SHARED_LINKER_FLAGS "${CMAKE_SHARED_LINKER_FLAGS} -lc++abi")
|
||||
endif(SIMDJSON_USE_LIBCPP)
|
||||
|
||||
# prevent shared libraries from depending on Intel provided libraries
|
||||
if(${CMAKE_C_COMPILER_ID} MATCHES "Intel") # icc / icpc
|
||||
set(CMAKE_SHARED_LINKER_FLAGS "${CMAKE_SHARED_LINKER_FLAGS} -static-intel")
|
||||
endif()
|
||||
|
||||
install(TARGETS simdjson-flags EXPORT simdjson-config)
|
||||
install(TARGETS simdjson-internal-flags EXPORT simdjson-config)
|
||||
@@ -0,0 +1,24 @@
|
||||
#
|
||||
# ${SIMDJSON_USER_CMAKECACHE} contains the *user-specified* simdjson options so you can call cmake on
|
||||
# another branch or repository with the same options.
|
||||
#
|
||||
# Not supported on Windows at present, because the only thing that uses it is checkperf, which we
|
||||
# don't run on Windows.
|
||||
#
|
||||
set(SIMDJSON_USER_CMAKECACHE ${CMAKE_CURRENT_BINARY_DIR}/.simdjson-user-CMakeCache.txt)
|
||||
if (MSVC)
|
||||
add_custom_command(
|
||||
OUTPUT ${SIMDJSON_USER_CMAKECACHE}
|
||||
COMMAND findstr SIMDJSON_ ${PROJECT_BINARY_DIR}/CMakeCache.txt > ${SIMDJSON_USER_CMAKECACHE}.tmp
|
||||
COMMAND findstr /v SIMDJSON_LIB_ ${SIMDJSON_USER_CMAKECACHE}.tmp > ${SIMDJSON_USER_CMAKECACHE}
|
||||
VERBATIM # Makes it not do weird escaping with the command
|
||||
)
|
||||
else()
|
||||
add_custom_command(
|
||||
OUTPUT ${SIMDJSON_USER_CMAKECACHE}
|
||||
COMMAND grep SIMDJSON_ ${PROJECT_BINARY_DIR}/CMakeCache.txt > ${SIMDJSON_USER_CMAKECACHE}.tmp
|
||||
COMMAND grep -v SIMDJSON_LIB_ ${SIMDJSON_USER_CMAKECACHE}.tmp > ${SIMDJSON_USER_CMAKECACHE}
|
||||
VERBATIM # Makes it not do weird escaping with the command
|
||||
)
|
||||
endif()
|
||||
add_custom_target(simdjson-user-cmakecache DEPENDS ${SIMDJSON_USER_CMAKECACHE})
|
||||
@@ -0,0 +1 @@
|
||||
.cache/
|
||||
Vendored
+110
@@ -0,0 +1,110 @@
|
||||
include(import.cmake)
|
||||
|
||||
option(SIMDJSON_COMPETITION "Compile competitive benchmarks" ON)
|
||||
option(SIMDJSON_GOOGLE_BENCHMARKS "compile the Google Benchmark benchmarks" ON)
|
||||
|
||||
if(SIMDJSON_GOOGLE_BENCHMARKS)
|
||||
set_off(BENCHMARK_ENABLE_TESTING)
|
||||
set_off(BENCHMARK_ENABLE_INSTALL)
|
||||
|
||||
import_dependency(google_benchmarks google/benchmark 8982e1e)
|
||||
add_dependency(google_benchmarks)
|
||||
endif()
|
||||
|
||||
# This prevents variables declared with set() from unnecessarily escaping and
|
||||
# should not be called more than once
|
||||
function(competition_scope_)
|
||||
# boost json in standalone mode requires C++17 string_view
|
||||
include(CheckCXXSourceCompiles)
|
||||
check_cxx_source_compiles("#include <string_view>\n#if __cpp_lib_string_view < 201606\n#error no string view support\n#endif\nint main(){}" USE_BOOST_JSON)
|
||||
if(USE_BOOST_JSON)
|
||||
import_dependency(boostjson boostorg/json ee8d72d8502b409b5561200299cad30ccdb91415)
|
||||
add_library(boostjson STATIC "${boostjson_SOURCE_DIR}/src/src.cpp")
|
||||
target_compile_definitions(boostjson PUBLIC BOOST_JSON_STANDALONE)
|
||||
target_include_directories(boostjson SYSTEM PUBLIC
|
||||
"${boostjson_SOURCE_DIR}/include")
|
||||
endif()
|
||||
|
||||
import_dependency(cjson DaveGamble/cJSON c69134d)
|
||||
add_library(cjson STATIC "${cjson_SOURCE_DIR}/cJSON.c")
|
||||
target_include_directories(cjson SYSTEM PUBLIC "${cjson_SOURCE_DIR}")
|
||||
|
||||
import_dependency(fastjson mikeando/fastjson 485f994)
|
||||
add_library(fastjson STATIC
|
||||
"${fastjson_SOURCE_DIR}/src/fastjson.cpp"
|
||||
"${fastjson_SOURCE_DIR}/src/fastjson2.cpp"
|
||||
"${fastjson_SOURCE_DIR}/src/fastjson_dom.cpp")
|
||||
target_include_directories(fastjson SYSTEM PUBLIC
|
||||
"${fastjson_SOURCE_DIR}/include")
|
||||
|
||||
import_dependency(gason vivkin/gason 7aee524)
|
||||
add_library(gason STATIC "${gason_SOURCE_DIR}/src/gason.cpp")
|
||||
target_include_directories(gason SYSTEM PUBLIC "${gason_SOURCE_DIR}/src")
|
||||
|
||||
import_dependency(jsmn zserge/jsmn 18e9fe4)
|
||||
add_library(jsmn STATIC "${jsmn_SOURCE_DIR}/jsmn.c")
|
||||
target_include_directories(jsmn SYSTEM PUBLIC "${jsmn_SOURCE_DIR}")
|
||||
|
||||
message(STATUS "Importing json (nlohmann/json@v3.9.1)")
|
||||
set(nlohmann_json_SOURCE_DIR "${dep_root}/json")
|
||||
if(NOT EXISTS "${nlohmann_json_SOURCE_DIR}")
|
||||
file(DOWNLOAD
|
||||
"https://github.com/nlohmann/json/releases/download/v3.9.1/json.hpp"
|
||||
"${nlohmann_json_SOURCE_DIR}/nlohmann/json.hpp")
|
||||
endif()
|
||||
add_library(nlohmann_json INTERFACE)
|
||||
target_include_directories(nlohmann_json SYSTEM INTERFACE "${nlohmann_json_SOURCE_DIR}")
|
||||
|
||||
import_dependency(json11 dropbox/json11 ec4e452)
|
||||
add_library(json11 STATIC "${json11_SOURCE_DIR}/json11.cpp")
|
||||
target_include_directories(json11 SYSTEM PUBLIC "${json11_SOURCE_DIR}")
|
||||
|
||||
set(jsoncpp_SOURCE_DIR "${simdjson_SOURCE_DIR}/dependencies/jsoncppdist")
|
||||
add_library(jsoncpp STATIC "${jsoncpp_SOURCE_DIR}/jsoncpp.cpp")
|
||||
target_include_directories(jsoncpp SYSTEM PUBLIC "${jsoncpp_SOURCE_DIR}")
|
||||
|
||||
import_dependency(rapidjson Tencent/rapidjson b32cd94)
|
||||
add_library(rapidjson INTERFACE)
|
||||
target_compile_definitions(rapidjson INTERFACE RAPIDJSON_HAS_STDSTRING)
|
||||
target_include_directories(rapidjson SYSTEM INTERFACE
|
||||
"${rapidjson_SOURCE_DIR}/include")
|
||||
|
||||
import_dependency(sajson chadaustin/sajson 2dcfd35)
|
||||
add_library(sajson INTERFACE)
|
||||
target_compile_definitions(sajson INTERFACE SAJSON_UNSORTED_OBJECT_KEYS)
|
||||
target_include_directories(sajson SYSTEM INTERFACE
|
||||
"${sajson_SOURCE_DIR}/include")
|
||||
|
||||
import_dependency(ujson4c esnme/ujson4c e14f3fd)
|
||||
add_library(ujson4c STATIC
|
||||
"${ujson4c_SOURCE_DIR}/src/ujdecode.c"
|
||||
"${ujson4c_SOURCE_DIR}/3rdparty/ultrajsondec.c")
|
||||
target_include_directories(ujson4c SYSTEM PUBLIC
|
||||
"${ujson4c_SOURCE_DIR}/src"
|
||||
"${ujson4c_SOURCE_DIR}/3rdparty")
|
||||
|
||||
import_dependency(yyjson ibireme/yyjson aa33ec5)
|
||||
add_library(yyjson STATIC "${yyjson_SOURCE_DIR}/src/yyjson.c")
|
||||
target_include_directories(yyjson SYSTEM PUBLIC "${yyjson_SOURCE_DIR}/src")
|
||||
|
||||
add_library(competition-core INTERFACE)
|
||||
target_link_libraries(competition-core INTERFACE nlohmann_json rapidjson sajson cjson jsmn yyjson)
|
||||
if(USE_BOOST_JSON)
|
||||
target_compile_definitions(boostjson INTERFACE HAS_BOOST_JSON)
|
||||
target_link_libraries(competition-core INTERFACE boostjson)
|
||||
endif()
|
||||
|
||||
add_library(competition-all INTERFACE)
|
||||
target_link_libraries(competition-all INTERFACE competition-core jsoncpp json11 fastjson gason ujson4c)
|
||||
endfunction()
|
||||
|
||||
if(SIMDJSON_COMPETITION)
|
||||
competition_scope_()
|
||||
endif()
|
||||
|
||||
set_off(CXXOPTS_BUILD_EXAMPLES)
|
||||
set_off(CXXOPTS_BUILD_TESTS)
|
||||
set_off(CXXOPTS_ENABLE_INSTALL)
|
||||
|
||||
import_dependency(cxxopts jarro2783/cxxopts 794c975)
|
||||
add_dependency(cxxopts)
|
||||
Vendored
-1
Submodule dependencies/cJSON deleted from c69134d017
Vendored
-1
Submodule dependencies/fastjson deleted from 485f994a61
Vendored
-1
Submodule dependencies/gason deleted from 7aee524189
Vendored
+48
@@ -0,0 +1,48 @@
|
||||
set(dep_root "${simdjson_SOURCE_DIR}/dependencies/.cache")
|
||||
if(DEFINED ENV{simdjson_DEPENDENCY_CACHE_DIR})
|
||||
set(dep_root "$ENV{simdjson_DEPENDENCY_CACHE_DIR}")
|
||||
endif()
|
||||
|
||||
function(import_dependency NAME GITHUB_REPO COMMIT)
|
||||
message(STATUS "Importing ${NAME} (${GITHUB_REPO}@${COMMIT})")
|
||||
set(target "${dep_root}/${NAME}")
|
||||
|
||||
# If the folder exists in the cache, then we assume that everything is as
|
||||
# should be and do nothing
|
||||
if(EXISTS "${target}")
|
||||
set("${NAME}_SOURCE_DIR" "${target}" PARENT_SCOPE)
|
||||
return()
|
||||
endif()
|
||||
|
||||
set(zip_url "https://github.com/${GITHUB_REPO}/archive/${COMMIT}.zip")
|
||||
set(archive "${dep_root}/archive.zip")
|
||||
set(dest "${dep_root}/_extract")
|
||||
|
||||
file(DOWNLOAD "${zip_url}" "${archive}")
|
||||
file(MAKE_DIRECTORY "${dest}")
|
||||
execute_process(
|
||||
WORKING_DIRECTORY "${dest}"
|
||||
COMMAND "${CMAKE_COMMAND}" -E tar xf "${archive}")
|
||||
file(REMOVE "${archive}")
|
||||
|
||||
# GitHub archives only ever have one folder component at the root, so this
|
||||
# will always match that single folder
|
||||
file(GLOB dir LIST_DIRECTORIES YES "${dest}/*")
|
||||
|
||||
file(RENAME "${dir}" "${target}")
|
||||
|
||||
set("${NAME}_SOURCE_DIR" "${target}" PARENT_SCOPE)
|
||||
endfunction()
|
||||
|
||||
# Delegates to the dependency
|
||||
macro(add_dependency NAME)
|
||||
if(NOT DEFINED "${NAME}_SOURCE_DIR")
|
||||
message(FATAL_ERROR "Missing ${NAME}_SOURCE_DIR variable")
|
||||
endif()
|
||||
|
||||
add_subdirectory("${${NAME}_SOURCE_DIR}" "${PROJECT_BINARY_DIR}/_deps/${NAME}")
|
||||
endmacro()
|
||||
|
||||
function(set_off NAME)
|
||||
set("${NAME}" OFF CACHE INTERNAL "")
|
||||
endfunction()
|
||||
Vendored
-1
Submodule dependencies/jsmn deleted from 18e9fe42cb
Vendored
-1
Submodule dependencies/json11 deleted from ec4e45219a
Vendored
-1
Submodule dependencies/jsoncpp deleted from 0c1cc6e1a3
+7
-7
@@ -7,28 +7,28 @@
|
||||
// //////////////////////////////////////////////////////////////////////
|
||||
|
||||
/*
|
||||
The JsonCpp library's source code, including accompanying documentation,
|
||||
The JsonCpp library's source code, including accompanying documentation,
|
||||
tests and demonstration applications, are licensed under the following
|
||||
conditions...
|
||||
|
||||
Baptiste Lepilleur and The JsonCpp Authors explicitly disclaim copyright in all
|
||||
jurisdictions which recognize such a disclaimer. In such jurisdictions,
|
||||
Baptiste Lepilleur and The JsonCpp Authors explicitly disclaim copyright in all
|
||||
jurisdictions which recognize such a disclaimer. In such jurisdictions,
|
||||
this software is released into the Public Domain.
|
||||
|
||||
In jurisdictions which do not recognize Public Domain property (e.g. Germany as of
|
||||
2010), this software is Copyright (c) 2007-2010 by Baptiste Lepilleur and
|
||||
The JsonCpp Authors, and is released under the terms of the MIT License (see below).
|
||||
|
||||
In jurisdictions which recognize Public Domain property, the user of this
|
||||
software may choose to accept it either as 1) Public Domain, 2) under the
|
||||
conditions of the MIT License (see below), or 3) under the terms of dual
|
||||
In jurisdictions which recognize Public Domain property, the user of this
|
||||
software may choose to accept it either as 1) Public Domain, 2) under the
|
||||
conditions of the MIT License (see below), or 3) under the terms of dual
|
||||
Public Domain/MIT License conditions described here, as they choose.
|
||||
|
||||
The MIT License is about as close to Public Domain as a license can get, and is
|
||||
described in clear, concise terms at:
|
||||
|
||||
http://en.wikipedia.org/wiki/MIT_License
|
||||
|
||||
|
||||
The full text of the MIT License follows:
|
||||
|
||||
========================================================================
|
||||
|
||||
Vendored
+7
-7
@@ -6,28 +6,28 @@
|
||||
// //////////////////////////////////////////////////////////////////////
|
||||
|
||||
/*
|
||||
The JsonCpp library's source code, including accompanying documentation,
|
||||
The JsonCpp library's source code, including accompanying documentation,
|
||||
tests and demonstration applications, are licensed under the following
|
||||
conditions...
|
||||
|
||||
Baptiste Lepilleur and The JsonCpp Authors explicitly disclaim copyright in all
|
||||
jurisdictions which recognize such a disclaimer. In such jurisdictions,
|
||||
Baptiste Lepilleur and The JsonCpp Authors explicitly disclaim copyright in all
|
||||
jurisdictions which recognize such a disclaimer. In such jurisdictions,
|
||||
this software is released into the Public Domain.
|
||||
|
||||
In jurisdictions which do not recognize Public Domain property (e.g. Germany as of
|
||||
2010), this software is Copyright (c) 2007-2010 by Baptiste Lepilleur and
|
||||
The JsonCpp Authors, and is released under the terms of the MIT License (see below).
|
||||
|
||||
In jurisdictions which recognize Public Domain property, the user of this
|
||||
software may choose to accept it either as 1) Public Domain, 2) under the
|
||||
conditions of the MIT License (see below), or 3) under the terms of dual
|
||||
In jurisdictions which recognize Public Domain property, the user of this
|
||||
software may choose to accept it either as 1) Public Domain, 2) under the
|
||||
conditions of the MIT License (see below), or 3) under the terms of dual
|
||||
Public Domain/MIT License conditions described here, as they choose.
|
||||
|
||||
The MIT License is about as close to Public Domain as a license can get, and is
|
||||
described in clear, concise terms at:
|
||||
|
||||
http://en.wikipedia.org/wiki/MIT_License
|
||||
|
||||
|
||||
The full text of the MIT License follows:
|
||||
|
||||
========================================================================
|
||||
|
||||
Vendored
+15
-7
@@ -6,28 +6,28 @@
|
||||
// //////////////////////////////////////////////////////////////////////
|
||||
|
||||
/*
|
||||
The JsonCpp library's source code, including accompanying documentation,
|
||||
The JsonCpp library's source code, including accompanying documentation,
|
||||
tests and demonstration applications, are licensed under the following
|
||||
conditions...
|
||||
|
||||
Baptiste Lepilleur and The JsonCpp Authors explicitly disclaim copyright in all
|
||||
jurisdictions which recognize such a disclaimer. In such jurisdictions,
|
||||
Baptiste Lepilleur and The JsonCpp Authors explicitly disclaim copyright in all
|
||||
jurisdictions which recognize such a disclaimer. In such jurisdictions,
|
||||
this software is released into the Public Domain.
|
||||
|
||||
In jurisdictions which do not recognize Public Domain property (e.g. Germany as of
|
||||
2010), this software is Copyright (c) 2007-2010 by Baptiste Lepilleur and
|
||||
The JsonCpp Authors, and is released under the terms of the MIT License (see below).
|
||||
|
||||
In jurisdictions which recognize Public Domain property, the user of this
|
||||
software may choose to accept it either as 1) Public Domain, 2) under the
|
||||
conditions of the MIT License (see below), or 3) under the terms of dual
|
||||
In jurisdictions which recognize Public Domain property, the user of this
|
||||
software may choose to accept it either as 1) Public Domain, 2) under the
|
||||
conditions of the MIT License (see below), or 3) under the terms of dual
|
||||
Public Domain/MIT License conditions described here, as they choose.
|
||||
|
||||
The MIT License is about as close to Public Domain as a license can get, and is
|
||||
described in clear, concise terms at:
|
||||
|
||||
http://en.wikipedia.org/wiki/MIT_License
|
||||
|
||||
|
||||
The full text of the MIT License follows:
|
||||
|
||||
========================================================================
|
||||
@@ -2652,10 +2652,18 @@ char const* Exception::what() const JSONCPP_NOEXCEPT { return msg_.c_str(); }
|
||||
RuntimeError::RuntimeError(String const& msg) : Exception(msg) {}
|
||||
LogicError::LogicError(String const& msg) : Exception(msg) {}
|
||||
JSONCPP_NORETURN void throwRuntimeError(String const& msg) {
|
||||
#if __cpp_exceptions
|
||||
throw RuntimeError(msg);
|
||||
#else
|
||||
abort();
|
||||
#endif
|
||||
}
|
||||
JSONCPP_NORETURN void throwLogicError(String const& msg) {
|
||||
#if __cpp_exceptions
|
||||
throw LogicError(msg);
|
||||
#else
|
||||
abort();
|
||||
#endif
|
||||
}
|
||||
|
||||
// //////////////////////////////////////////////////////////////////
|
||||
|
||||
Vendored
-1
Submodule dependencies/rapidjson deleted from b32cd9421c
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user