File: json0.py 1 #!/usr/bin/python 2 3 # The MIT License (MIT) 4 # 5 # Copyright (c) 2026 pacman64 6 # 7 # Permission is hereby granted, free of charge, to any person obtaining a copy 8 # of this software and associated documentation files (the "Software"), to deal 9 # in the Software without restriction, including without limitation the rights 10 # to use, copy, modify, merge, publish, distribute, sublicense, and/or sell 11 # copies of the Software, and to permit persons to whom the Software is 12 # furnished to do so, subject to the following conditions: 13 # 14 # The above copyright notice and this permission notice shall be included in 15 # all copies or substantial portions of the Software. 16 # 17 # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR 18 # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, 19 # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE 20 # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER 21 # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, 22 # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE 23 # SOFTWARE. 24 25 26 from io import BufferedReader, BytesIO 27 from sys import argv, exit, stderr, stdin, stdout 28 29 30 info = ''' 31 json0 [options...] [filepath/URI...] 32 33 34 JSON-0 converts/fixes JSON/pseudo-JSON input into minimal JSON output. 35 36 Besides minimizing bytes, this tool also adapts almost-JSON input into valid 37 JSON, since it ignores comments and trailing commas, neither of which are 38 supported in JSON, but which are still commonly used. 39 40 It also turns single-quoted strings into proper double-quoted ones, as well 41 as change invalid 2-digit `\\x` hexadecimal escapes into JSON's 4-digit `\\u` 42 hexadecimal escapes. When backslashes in strings are followed by an invalid 43 escape letter, the backslash is ignored. 44 45 Output is always a single line of valid JSON, ending with a line-feed. 46 47 All (optional) leading options start with either single or double-dash: 48 49 -h, -help show this help message 50 ''' 51 52 # handle standard help cmd-line options, quitting right away in that case 53 if len(argv) > 1 and argv[1] in ('-h', '--h', '-help', '--help'): 54 print(info.strip()) 55 exit(0) 56 57 58 # note: using regexes doesn't seem to speed-up number/string-handling 59 60 61 def read(r, size: int) -> bytes: 62 global pos, linenum 63 64 chunk = r.read(size) 65 if not chunk: 66 return chunk 67 68 if not (10 in chunk): 69 pos += len(chunk) 70 return chunk 71 72 for b in chunk: 73 if b == 10: 74 pos = 1 75 linenum += 1 76 else: 77 pos += 1 78 return chunk 79 80 81 def skip_byte(r) -> None: 82 global pos, linenum 83 84 chunk = r.read(1) 85 if not chunk: 86 return 87 88 if chunk[0] == 10: 89 pos = 1 90 linenum += 1 91 else: 92 pos += 1 93 94 95 def peek_byte(r) -> int: 96 chunk = r.peek(64) 97 if len(chunk) > 0: 98 return chunk[0] 99 return -1 100 101 102 def handle_array(w, r) -> None: 103 seek_next = seek_next_token 104 105 n = 0 106 lead = peek_byte(r) 107 end = 0 108 if lead < 0: 109 raise ValueError('unexpected end of input data, before "]"') 110 if lead == 91: # ord('[') 111 end = 93 # ord(']') 112 elif lead == 40: # ord('(') 113 end = 41 # ord(')') 114 else: 115 raise ValueError('expected "[" or "("') 116 skip_byte(r) 117 w.write(b'[') 118 119 while True: 120 # whitespace/comments may precede the next item/comma 121 seek_next(r) 122 b = peek_byte(r) 123 if b < 0: 124 raise ValueError('unexpected end of input data, before "]"') 125 126 comma = b == 44 # ord(',') 127 128 if comma: 129 skip_byte(r) 130 # whitespace/comments may follow the comma 131 seek_next(r) 132 b = peek_byte(r) 133 if b < 0: 134 raise ValueError('unexpected end of input data, before "]"') 135 136 if b == end: 137 skip_byte(r) 138 w.write(b']') 139 return 140 141 if n > 0: 142 if not comma: 143 raise ValueError('missing a comma between array values') 144 w.write(b',') 145 146 b = peek_byte(r) 147 if b > 0: 148 handlers[b](w, r) 149 n += 1 150 151 152 def handle_double_quoted_string(w, r) -> None: 153 skip_byte(r) 154 w.write(b'"') 155 handle_inner_string(w, r, 34) # ord('"') 156 w.write(b'"') 157 158 159 def handle_dot(w, r) -> None: 160 skip_byte(r) 161 # precede the leading decimal dot with a 0 162 w.write(b'0.') 163 164 # handle decimals, which in this case aren't optional, as a leading 165 # dot is what led to this point 166 if copy_digits(w, r) < 1: 167 raise ValueError('expected numeric digits, but found none') 168 169 170 def handle_false(w, r) -> None: 171 demand(r, b'false') 172 w.write(b'false') 173 174 175 def handle_False(w, r) -> None: 176 demand(r, b'False') 177 w.write(b'false') 178 179 180 def handle_invalid(w, r) -> None: 181 b = peek_byte(r) 182 if b < 0: 183 raise ValueError('unexpected end of input data') 184 # raise ValueError(f'unexpected JSON byte-value {b}') 185 if 32 < b <= 126: 186 msg = f'unexpected symbol {chr(b)}' 187 else: 188 msg = f'unexpected byte-value {b}' 189 raise ValueError(msg) 190 191 192 def handle_negative(w, r) -> None: 193 skip_byte(r) 194 w.write(b'-') 195 196 if peek_byte(r) == 46: # ord('.') 197 skip_byte(r) 198 w.write(b'0.') 199 if copy_digits(w, r) < 1: 200 raise ValueError('expected numeric digits, but found none') 201 else: 202 handle_number(w, r) 203 204 205 def handle_null(w, r) -> None: 206 demand(r, b'null') 207 w.write(b'null') 208 209 210 def handle_None(w, r) -> None: 211 demand(r, b'None') 212 w.write(b'null') 213 214 215 def handle_number(w, r) -> None: 216 # handle integer part 217 if copy_digits(w, r) < 1: 218 raise ValueError('expected numeric digits, but found none') 219 # ignore optional trailing 'n', used in javascript bigint literals 220 if peek_byte(r) == 110: # ord('n') 221 skip_byte(r) 222 return 223 224 # handle optional decimals 225 b = peek_byte(r) 226 if b == 46: # ord('.') 227 skip_byte(r) 228 w.write(b'.') 229 if copy_digits(w, r) < 1: 230 # follow a trailing decimal dot with a 0 231 w.write(b'0') 232 233 # handle optional exponent 234 if b == 101 or b == 69: # ord('e'), ord('E') 235 skip_byte(r) 236 w.write(b'e' if b == 101 else b'E') 237 b = peek_byte(r) 238 if b == 43: # ord('+') 239 skip_byte(r) 240 elif b == 45: # ord('-') 241 w.write(b'-') 242 skip_byte(r) 243 if copy_digits(w, r) < 1: 244 raise ValueError('expected numeric digits, but found none') 245 246 247 def handle_object(w, r) -> None: 248 seek_next = seek_next_token 249 250 num_pairs = 0 251 skip_byte(r) 252 w.write(b'{') 253 254 while True: 255 # whitespace/comments may precede the next item/comma 256 seek_next(r) 257 b = peek_byte(r) 258 if b < 0: 259 raise ValueError('unexpected end of input data, before "}"') 260 261 comma = b == 44 # ord(',') 262 263 if comma: 264 skip_byte(r) 265 # whitespace/comments may follow the comma 266 seek_next(r) 267 b = peek_byte(r) 268 if b < 0: 269 raise ValueError('unexpected end of input data, before "}"') 270 271 if b == 125: # ord('}') 272 skip_byte(r) 273 w.write(b'}') 274 return 275 276 if num_pairs > 0: 277 if not comma: 278 raise ValueError('missing a comma between key-value pairs') 279 w.write(b',') 280 281 demand_string(w, r) 282 # whitespace/comments may follow the key 283 seek_next(r) 284 demand(r, b':') 285 w.write(b':') 286 # whitespace/comments may follow the colon 287 seek_next(r) 288 b = peek_byte(r) 289 if b > 0: 290 handlers[b](w, r) 291 num_pairs += 1 292 293 294 def handle_positive(w, r) -> None: 295 # do nothing with the leading plus sign, which isn't allowed in JSON 296 skip_byte(r) 297 298 if peek_byte(r) == 46: # ord('.') 299 skip_byte(r) 300 w.write(b'0.') 301 if copy_digits(w, r) < 1: 302 raise ValueError('expected numeric digits, but found none') 303 else: 304 handle_number(w, r) 305 306 307 def handle_single_quoted_string(w, r) -> None: 308 skip_byte(r) 309 w.write(b'"') 310 handle_inner_string(w, r, 39) # ord('\'') 311 w.write(b'"') 312 313 314 def demand_string(w, r) -> None: 315 quote = peek_byte(r) 316 if quote < 0: 317 msg = 'unexpected end of input, instead of a string quote' 318 raise ValueError(msg) 319 320 if quote == 34: # ord('"') 321 handle_double_quoted_string(w, r) 322 return 323 324 if quote == 39: # ord('\'') 325 handle_single_quoted_string(w, r) 326 return 327 328 if 32 < quote <= 126: # ord(' '), ord('~') 329 msg = f'expected ", or even \', but got "{chr(quote)}" instead' 330 else: 331 msg = f'expected ", or even \', but got byte "{quote}" instead' 332 raise ValueError(msg) 333 334 335 def handle_inner_string(w, r, quote: int) -> None: 336 esc = False 337 bad_hex_msg = 'invalid hexadecimal symbols' 338 early_end_msg = 'input data ended while still in quoted string' 339 340 def is_hex(x: int) -> bool: 341 # 48 is ord('0'), 57 is ord('9'), 97 is ord('a'), 102 is ord('f') 342 return 48 <= x <= 57 or 97 <= x <= 102 343 344 def lower(x: int) -> bool: 345 # 65 is ord('A'), 90 is ord('Z') 346 return x + 32 if 65 <= x <= 90 else x 347 348 while True: 349 chunk = r.peek(1) 350 if len(chunk) < 1: 351 raise ValueError(early_end_msg) 352 b = chunk[0] 353 354 if esc: 355 esc = False 356 357 if b == 120: # ord('x') 358 skip_byte(r) 359 chunk = read(r, 2) 360 if len(chunk) != 2: 361 raise ValueError(early_end_msg) 362 a = lower(chunk[0]) 363 b = lower(chunk[1]) 364 w.write(b'\\u00') 365 if not (is_hex(a) and is_hex(b)): 366 raise ValueError(bad_hex_msg) 367 w.write(a) 368 w.write(b) 369 continue 370 371 if b == 117: # ord('u') 372 skip_byte(r) 373 chunk = read(r, 4) 374 if len(chunk) != 4: 375 raise ValueError(early_end_msg) 376 a = lower(chunk[0]) 377 b = lower(chunk[1]) 378 c = lower(chunk[2]) 379 d = lower(chunk[3]) 380 w.write(b'\\u') 381 if not (is_hex(a) and is_hex(b) and is_hex(c) and is_hex(d)): 382 raise ValueError(bad_hex_msg) 383 w.write(chunk) 384 continue 385 386 # numbers for '"', '\\', 'n', 't', 'r', 'b', and 'f' 387 if b in (34, 92, 110, 116, 114, 98, 102): 388 w.write(b'\\') 389 390 w.write(read(r, 1)) 391 continue 392 393 if b == 92: # ord('\\') 394 esc = True 395 skip_byte(r) 396 continue 397 398 if b == quote: 399 skip_byte(r) 400 return 401 402 # emit normal string-byte 403 w.write(read(r, 1)) 404 405 406 def handle_true(w, r) -> None: 407 demand(r, b'true') 408 w.write(b'true') 409 410 411 def handle_True(w, r) -> None: 412 demand(r, b'True') 413 w.write(b'true') 414 415 416 # setup byte-handling lookup tuple 417 bh = [handle_invalid for i in range(256)] 418 bh[ord('0')] = handle_number 419 bh[ord('1')] = handle_number 420 bh[ord('2')] = handle_number 421 bh[ord('3')] = handle_number 422 bh[ord('4')] = handle_number 423 bh[ord('5')] = handle_number 424 bh[ord('6')] = handle_number 425 bh[ord('7')] = handle_number 426 bh[ord('8')] = handle_number 427 bh[ord('9')] = handle_number 428 bh[ord('+')] = handle_positive 429 bh[ord('-')] = handle_negative 430 bh[ord('.')] = handle_dot 431 bh[ord('"')] = handle_double_quoted_string 432 bh[ord('\'')] = handle_single_quoted_string 433 bh[ord('F')] = handle_False 434 bh[ord('N')] = handle_None 435 bh[ord('T')] = handle_True 436 bh[ord('f')] = handle_false 437 bh[ord('n')] = handle_null 438 bh[ord('t')] = handle_true 439 bh[ord('[')] = handle_array 440 bh[ord('(')] = handle_array 441 bh[ord('{')] = handle_object 442 443 # handlers is the immutable byte-driven func-dispatch table 444 handlers = tuple(bh) 445 446 447 def copy_digits(w, r) -> int: 448 'Returns how many digits were copied/handled.' 449 450 copied = 0 451 while True: 452 chunk = r.peek(64) 453 if len(chunk) == 0: 454 return copied 455 456 i = find_digits_end_index(chunk) 457 if i >= 0: 458 w.write(read(r, i)) 459 copied += i 460 return copied 461 else: 462 w.write(chunk) 463 read(r, len(chunk)) 464 copied += len(chunk) 465 466 467 def seek_next_token(r) -> None: 468 'Skip an arbitrarily-long mix of whitespace and comments.' 469 470 while True: 471 chunk = r.peek(1024) 472 if len(chunk) == 0: 473 # input is over, and this func doesn't consider that an error 474 return 475 476 comment = False 477 478 for i, b in enumerate(chunk): 479 # skip space, tab, line-feed, carriage-return, or form-feed 480 if b in (9, 10, 11, 13, 32): 481 continue 482 483 if b == 47 or b == 35: # ord('/'), ord('#') 484 read(r, i) 485 demand_comment(r) 486 comment = True 487 break 488 489 # found start of next token 490 read(r, i) 491 return 492 493 if not comment: 494 read(r, len(chunk)) 495 496 497 def skip_line(r) -> None: 498 while True: 499 chunk = r.peek(1024) 500 if len(chunk) == 0: 501 return 502 503 i = chunk.find(b'\n') 504 if i >= 0: 505 read(r, i + 1) 506 return 507 508 read(r, len(chunk)) 509 510 511 def skip_general_comment(r) -> None: 512 while True: 513 chunk = r.peek(1024) 514 if len(chunk) == 0: 515 raise ValueError(f'input data ended before an expected */') 516 517 i = chunk.find(b'*') 518 if i < 0: 519 # no */ in this chunk, so skip it and try with the next one 520 read(r, len(chunk)) 521 continue 522 523 # skip right past the * just found, then check if a / follows it 524 read(r, i + 1) 525 if peek_byte(r) == 47: # ord('/') 526 # got */, the end of this comment 527 skip_byte(r) 528 return 529 530 531 def find_digits_end_index(chunk: bytes) -> int: 532 i = 0 533 for b in chunk: 534 if 48 <= b <= 57: 535 i += 1 536 else: 537 return i 538 539 # all bytes (if any) were digits, so no end was found 540 return -1 541 542 543 def demand(r, what: bytes) -> None: 544 lead = read(r, len(what)) 545 if not lead.startswith(what): 546 lead = str(lead, encoding='utf-8') 547 what = str(what, encoding='utf-8') 548 raise ValueError(f'expected {what}, but got {lead} instead') 549 550 551 def demand_comment(r) -> None: 552 b = peek_byte(r) 553 if b < 0: 554 raise ValueError('unexpected end of input data') 555 if b == 35: # ord('#') 556 # handle single-line comment 557 skip_line(r) 558 return 559 560 demand(r, b'/') 561 b = peek_byte(r) 562 if b < 0: 563 raise ValueError('unexpected end of input data') 564 565 if b == 47: # ord('/') 566 # handle single-line comment 567 skip_line(r) 568 return 569 570 if b == 42: # ord('*') 571 # handle (potentially) multi-line comment 572 skip_general_comment(r) 573 return 574 575 raise ValueError('expected * or another /, after a /') 576 577 578 def json0(w, src, end) -> None: 579 r = BufferedReader(src) 580 581 # skip leading UTF-8 BOM (byte-order mark) 582 if r.peek(3) == b'\xef\xbb\xbf': 583 read(r, 3) 584 585 # skip leading whitespace/comments 586 seek_next_token(r) 587 588 # emit a single output line, ending with a line-feed 589 b = peek_byte(r) 590 if b >= 0: 591 handlers[b](w, r) 592 else: 593 # w.write(b'null') 594 # treat empty(ish) input as invalid JSON 595 raise ValueError('can\'t turn empty(ish) input into JSON') 596 597 # deliberately run post-processing before checking for trailing-data 598 # errors: for example, if post-proc func emits new line, errors will 599 # show up on their separate line, which is nicer 600 end(w) 601 602 # ignore trailing whitespace/comment bytes, if present 603 seek_next_token(r) 604 605 # ignore trailing semicolon, if present 606 b = peek_byte(r) 607 if b == 59: # ord(';') 608 read(r, 1) 609 # ignore trailing whitespace/comment bytes, if present 610 seek_next_token(r) 611 612 if len(r.peek(1)) > 0: 613 raise ValueError('unexpected trailing bytes in JSON data') 614 615 616 def seems_url(s: str) -> bool: 617 protocols = ('https://', 'http://', 'file://', 'ftp://', 'data:') 618 return any(s.startswith(p) for p in protocols) 619 620 621 def handle_json(w, r) -> None: 622 def end(w) -> None: 623 w.write(b'\n') 624 w.flush() 625 json0(w, r, end) 626 627 628 def handle_json_lines(w, r) -> None: 629 global pos, linenum 630 631 items = 0 632 linenum = 0 633 w.write(b'[') 634 635 while True: 636 line = r.readline().lstrip() 637 if not line: 638 break 639 640 pos = 1 641 linenum += 1 642 643 stripped = line.strip() 644 if not stripped or stripped.startswith(b'//'): 645 continue 646 647 items += 1 648 if items > 1: 649 w.write(b',') 650 651 json0(w, BytesIO(line), lambda w: w.flush()) 652 653 w.write(b']\n') 654 655 656 start_args = 1 657 handle_input = handle_json 658 if len(argv) > 1 and argv[1] in ('-jl', '--jl', '-jsonl', '--jsonl'): 659 start_args = 2 660 handle_input = handle_json_lines 661 662 if len(argv) > start_args: 663 s = argv[start_args] 664 if s != '-' and s != '--' and s.startswith('-'): 665 print(f'json0: unsupported option {s}') 666 exit(1) 667 668 if len(argv) > start_args and argv[start_args] == '--': 669 start_args += 1 670 671 if len(argv) - 1 > start_args: 672 print(f'json0: multiple inputs not allowed', file=stderr) 673 exit(1) 674 675 w = stdout.buffer 676 name = argv[start_args] if len(argv) > start_args else '-' 677 678 # values keeping track of the input-position, shown in case of errors 679 pos = 1 680 linenum = 1 681 682 try: 683 if name == '-': 684 handle_input(w, stdin.buffer) 685 elif seems_url(name): 686 from urllib.request import urlopen 687 with urlopen(name) as inp: 688 handle_input(w, inp) 689 else: 690 with open(name, mode='rb') as inp: 691 handle_input(w, inp) 692 except BrokenPipeError: 693 # quit quietly, instead of showing a confusing error message 694 stderr.close() 695 exit(0) 696 except KeyboardInterrupt: 697 exit(2) 698 except Exception as e: 699 stdout.write('\n') 700 print(f'line {linenum}, pos {pos} : {e}', file=stderr) 701 exit(1)