File: json0.py
   1 #!/usr/bin/python
   2 
   3 # The MIT License (MIT)
   4 #
   5 # Copyright (c) 2026 pacman64
   6 #
   7 # Permission is hereby granted, free of charge, to any person obtaining a copy
   8 # of this software and associated documentation files (the "Software"), to deal
   9 # in the Software without restriction, including without limitation the rights
  10 # to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
  11 # copies of the Software, and to permit persons to whom the Software is
  12 # furnished to do so, subject to the following conditions:
  13 #
  14 # The above copyright notice and this permission notice shall be included in
  15 # all copies or substantial portions of the Software.
  16 #
  17 # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
  18 # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
  19 # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
  20 # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
  21 # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
  22 # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
  23 # SOFTWARE.
  24 
  25 
  26 from io import BufferedReader, BytesIO
  27 from sys import argv, exit, stderr, stdin, stdout
  28 
  29 
  30 info = '''
  31 json0 [options...] [filepath/URI...]
  32 
  33 
  34 JSON-0 converts/fixes JSON/pseudo-JSON input into minimal JSON output.
  35 
  36 Besides minimizing bytes, this tool also adapts almost-JSON input into valid
  37 JSON, since it ignores comments and trailing commas, neither of which are
  38 supported in JSON, but which are still commonly used.
  39 
  40 It also turns single-quoted strings into proper double-quoted ones, as well
  41 as change invalid 2-digit `\\x` hexadecimal escapes into JSON's 4-digit `\\u`
  42 hexadecimal escapes. When backslashes in strings are followed by an invalid
  43 escape letter, the backslash is ignored.
  44 
  45 Output is always a single line of valid JSON, ending with a line-feed.
  46 
  47 All (optional) leading options start with either single or double-dash:
  48 
  49     -h, -help    show this help message
  50 '''
  51 
  52 # handle standard help cmd-line options, quitting right away in that case
  53 if len(argv) > 1 and argv[1] in ('-h', '--h', '-help', '--help'):
  54     print(info.strip())
  55     exit(0)
  56 
  57 
  58 # note: using regexes doesn't seem to speed-up number/string-handling
  59 
  60 
  61 def read(r, size: int) -> bytes:
  62     global pos, linenum
  63 
  64     chunk = r.read(size)
  65     if not chunk:
  66         return chunk
  67 
  68     if not (10 in chunk):
  69         pos += len(chunk)
  70         return chunk
  71 
  72     for b in chunk:
  73         if b == 10:
  74             pos = 1
  75             linenum += 1
  76         else:
  77             pos += 1
  78     return chunk
  79 
  80 
  81 def skip_byte(r) -> None:
  82     global pos, linenum
  83 
  84     chunk = r.read(1)
  85     if not chunk:
  86         return
  87 
  88     if chunk[0] == 10:
  89         pos = 1
  90         linenum += 1
  91     else:
  92         pos += 1
  93 
  94 
  95 def peek_byte(r) -> int:
  96     chunk = r.peek(64)
  97     if len(chunk) > 0:
  98         return chunk[0]
  99     return -1
 100 
 101 
 102 def handle_array(w, r) -> None:
 103     seek_next = seek_next_token
 104 
 105     n = 0
 106     lead = peek_byte(r)
 107     end = 0
 108     if lead < 0:
 109         raise ValueError('unexpected end of input data, before "]"')
 110     if lead == 91: # ord('[')
 111         end = 93 # ord(']')
 112     elif lead == 40: # ord('(')
 113         end = 41 # ord(')')
 114     else:
 115         raise ValueError('expected "[" or "("')
 116     skip_byte(r)
 117     w.write(b'[')
 118 
 119     while True:
 120         # whitespace/comments may precede the next item/comma
 121         seek_next(r)
 122         b = peek_byte(r)
 123         if b < 0:
 124             raise ValueError('unexpected end of input data, before "]"')
 125 
 126         comma = b == 44 # ord(',')
 127 
 128         if comma:
 129             skip_byte(r)
 130             # whitespace/comments may follow the comma
 131             seek_next(r)
 132             b = peek_byte(r)
 133             if b < 0:
 134                 raise ValueError('unexpected end of input data, before "]"')
 135 
 136         if b == end:
 137             skip_byte(r)
 138             w.write(b']')
 139             return
 140 
 141         if n > 0:
 142             if not comma:
 143                 raise ValueError('missing a comma between array values')
 144             w.write(b',')
 145 
 146         b = peek_byte(r)
 147         if b > 0:
 148             handlers[b](w, r)
 149             n += 1
 150 
 151 
 152 def handle_double_quoted_string(w, r) -> None:
 153     skip_byte(r)
 154     w.write(b'"')
 155     handle_inner_string(w, r, 34) # ord('"')
 156     w.write(b'"')
 157 
 158 
 159 def handle_dot(w, r) -> None:
 160     skip_byte(r)
 161     # precede the leading decimal dot with a 0
 162     w.write(b'0.')
 163 
 164     # handle decimals, which in this case aren't optional, as a leading
 165     # dot is what led to this point
 166     if copy_digits(w, r) < 1:
 167         raise ValueError('expected numeric digits, but found none')
 168 
 169 
 170 def handle_false(w, r) -> None:
 171     demand(r, b'false')
 172     w.write(b'false')
 173 
 174 
 175 def handle_False(w, r) -> None:
 176     demand(r, b'False')
 177     w.write(b'false')
 178 
 179 
 180 def handle_invalid(w, r) -> None:
 181     b = peek_byte(r)
 182     if b < 0:
 183         raise ValueError('unexpected end of input data')
 184     # raise ValueError(f'unexpected JSON byte-value {b}')
 185     if 32 < b <= 126:
 186         msg = f'unexpected symbol {chr(b)}'
 187     else:
 188         msg = f'unexpected byte-value {b}'
 189     raise ValueError(msg)
 190 
 191 
 192 def handle_negative(w, r) -> None:
 193     skip_byte(r)
 194     w.write(b'-')
 195 
 196     if peek_byte(r) == 46: # ord('.')
 197         skip_byte(r)
 198         w.write(b'0.')
 199         if copy_digits(w, r) < 1:
 200             raise ValueError('expected numeric digits, but found none')
 201     else:
 202         handle_number(w, r)
 203 
 204 
 205 def handle_null(w, r) -> None:
 206     demand(r, b'null')
 207     w.write(b'null')
 208 
 209 
 210 def handle_None(w, r) -> None:
 211     demand(r, b'None')
 212     w.write(b'null')
 213 
 214 
 215 def handle_number(w, r) -> None:
 216     # handle integer part
 217     if copy_digits(w, r) < 1:
 218         raise ValueError('expected numeric digits, but found none')
 219     # ignore optional trailing 'n', used in javascript bigint literals
 220     if peek_byte(r) == 110: # ord('n')
 221         skip_byte(r)
 222         return
 223 
 224     # handle optional decimals
 225     b = peek_byte(r)
 226     if b == 46: # ord('.')
 227         skip_byte(r)
 228         w.write(b'.')
 229         if copy_digits(w, r) < 1:
 230             # follow a trailing decimal dot with a 0
 231             w.write(b'0')
 232 
 233     # handle optional exponent
 234     if b == 101 or b == 69: # ord('e'), ord('E')
 235         skip_byte(r)
 236         w.write(b'e' if b == 101 else b'E')
 237         b = peek_byte(r)
 238         if b == 43: # ord('+')
 239             skip_byte(r)
 240         elif b == 45: # ord('-')
 241             w.write(b'-')
 242             skip_byte(r)
 243         if copy_digits(w, r) < 1:
 244             raise ValueError('expected numeric digits, but found none')
 245 
 246 
 247 def handle_object(w, r) -> None:
 248     seek_next = seek_next_token
 249 
 250     num_pairs = 0
 251     skip_byte(r)
 252     w.write(b'{')
 253 
 254     while True:
 255         # whitespace/comments may precede the next item/comma
 256         seek_next(r)
 257         b = peek_byte(r)
 258         if b < 0:
 259             raise ValueError('unexpected end of input data, before "}"')
 260 
 261         comma = b == 44 # ord(',')
 262 
 263         if comma:
 264             skip_byte(r)
 265             # whitespace/comments may follow the comma
 266             seek_next(r)
 267             b = peek_byte(r)
 268             if b < 0:
 269                 raise ValueError('unexpected end of input data, before "}"')
 270 
 271         if b == 125: # ord('}')
 272             skip_byte(r)
 273             w.write(b'}')
 274             return
 275 
 276         if num_pairs > 0:
 277             if not comma:
 278                 raise ValueError('missing a comma between key-value pairs')
 279             w.write(b',')
 280 
 281         demand_string(w, r)
 282         # whitespace/comments may follow the key
 283         seek_next(r)
 284         demand(r, b':')
 285         w.write(b':')
 286         # whitespace/comments may follow the colon
 287         seek_next(r)
 288         b = peek_byte(r)
 289         if b > 0:
 290             handlers[b](w, r)
 291             num_pairs += 1
 292 
 293 
 294 def handle_positive(w, r) -> None:
 295     # do nothing with the leading plus sign, which isn't allowed in JSON
 296     skip_byte(r)
 297 
 298     if peek_byte(r) == 46: # ord('.')
 299         skip_byte(r)
 300         w.write(b'0.')
 301         if copy_digits(w, r) < 1:
 302             raise ValueError('expected numeric digits, but found none')
 303     else:
 304         handle_number(w, r)
 305 
 306 
 307 def handle_single_quoted_string(w, r) -> None:
 308     skip_byte(r)
 309     w.write(b'"')
 310     handle_inner_string(w, r, 39) # ord('\'')
 311     w.write(b'"')
 312 
 313 
 314 def demand_string(w, r) -> None:
 315     quote = peek_byte(r)
 316     if quote < 0:
 317         msg = 'unexpected end of input, instead of a string quote'
 318         raise ValueError(msg)
 319 
 320     if quote == 34: # ord('"')
 321         handle_double_quoted_string(w, r)
 322         return
 323 
 324     if quote == 39: # ord('\'')
 325         handle_single_quoted_string(w, r)
 326         return
 327 
 328     if 32 < quote <= 126: # ord(' '), ord('~')
 329         msg = f'expected ", or even \', but got "{chr(quote)}" instead'
 330     else:
 331         msg = f'expected ", or even \', but got byte "{quote}" instead'
 332     raise ValueError(msg)
 333 
 334 
 335 def handle_inner_string(w, r, quote: int) -> None:
 336     esc = False
 337     bad_hex_msg = 'invalid hexadecimal symbols'
 338     early_end_msg = 'input data ended while still in quoted string'
 339 
 340     def is_hex(x: int) -> bool:
 341         # 48 is ord('0'), 57 is ord('9'), 97 is ord('a'), 102 is ord('f')
 342         return 48 <= x <= 57 or 97 <= x <= 102
 343 
 344     def lower(x: int) -> bool:
 345         # 65 is ord('A'), 90 is ord('Z')
 346         return x + 32 if 65 <= x <= 90 else x
 347 
 348     while True:
 349         chunk = r.peek(1)
 350         if len(chunk) < 1:
 351             raise ValueError(early_end_msg)
 352         b = chunk[0]
 353 
 354         if esc:
 355             esc = False
 356 
 357             if b == 120: # ord('x')
 358                 skip_byte(r)
 359                 chunk = read(r, 2)
 360                 if len(chunk) != 2:
 361                     raise ValueError(early_end_msg)
 362                 a = lower(chunk[0])
 363                 b = lower(chunk[1])
 364                 w.write(b'\\u00')
 365                 if not (is_hex(a) and is_hex(b)):
 366                     raise ValueError(bad_hex_msg)
 367                 w.write(a)
 368                 w.write(b)
 369                 continue
 370 
 371             if b == 117: # ord('u')
 372                 skip_byte(r)
 373                 chunk = read(r, 4)
 374                 if len(chunk) != 4:
 375                     raise ValueError(early_end_msg)
 376                 a = lower(chunk[0])
 377                 b = lower(chunk[1])
 378                 c = lower(chunk[2])
 379                 d = lower(chunk[3])
 380                 w.write(b'\\u')
 381                 if not (is_hex(a) and is_hex(b) and is_hex(c) and is_hex(d)):
 382                     raise ValueError(bad_hex_msg)
 383                 w.write(chunk)
 384                 continue
 385 
 386             # numbers for '"', '\\', 'n', 't', 'r', 'b', and 'f'
 387             if b in (34, 92, 110, 116, 114, 98, 102):
 388                 w.write(b'\\')
 389 
 390             w.write(read(r, 1))
 391             continue
 392 
 393         if b == 92: # ord('\\')
 394             esc = True
 395             skip_byte(r)
 396             continue
 397 
 398         if b == quote:
 399             skip_byte(r)
 400             return
 401 
 402         # emit normal string-byte
 403         w.write(read(r, 1))
 404 
 405 
 406 def handle_true(w, r) -> None:
 407     demand(r, b'true')
 408     w.write(b'true')
 409 
 410 
 411 def handle_True(w, r) -> None:
 412     demand(r, b'True')
 413     w.write(b'true')
 414 
 415 
 416 # setup byte-handling lookup tuple
 417 bh = [handle_invalid for i in range(256)]
 418 bh[ord('0')] = handle_number
 419 bh[ord('1')] = handle_number
 420 bh[ord('2')] = handle_number
 421 bh[ord('3')] = handle_number
 422 bh[ord('4')] = handle_number
 423 bh[ord('5')] = handle_number
 424 bh[ord('6')] = handle_number
 425 bh[ord('7')] = handle_number
 426 bh[ord('8')] = handle_number
 427 bh[ord('9')] = handle_number
 428 bh[ord('+')] = handle_positive
 429 bh[ord('-')] = handle_negative
 430 bh[ord('.')] = handle_dot
 431 bh[ord('"')] = handle_double_quoted_string
 432 bh[ord('\'')] = handle_single_quoted_string
 433 bh[ord('F')] = handle_False
 434 bh[ord('N')] = handle_None
 435 bh[ord('T')] = handle_True
 436 bh[ord('f')] = handle_false
 437 bh[ord('n')] = handle_null
 438 bh[ord('t')] = handle_true
 439 bh[ord('[')] = handle_array
 440 bh[ord('(')] = handle_array
 441 bh[ord('{')] = handle_object
 442 
 443 # handlers is the immutable byte-driven func-dispatch table
 444 handlers = tuple(bh)
 445 
 446 
 447 def copy_digits(w, r) -> int:
 448     'Returns how many digits were copied/handled.'
 449 
 450     copied = 0
 451     while True:
 452         chunk = r.peek(64)
 453         if len(chunk) == 0:
 454             return copied
 455 
 456         i = find_digits_end_index(chunk)
 457         if i >= 0:
 458             w.write(read(r, i))
 459             copied += i
 460             return copied
 461         else:
 462             w.write(chunk)
 463             read(r, len(chunk))
 464             copied += len(chunk)
 465 
 466 
 467 def seek_next_token(r) -> None:
 468     'Skip an arbitrarily-long mix of whitespace and comments.'
 469 
 470     while True:
 471         chunk = r.peek(1024)
 472         if len(chunk) == 0:
 473             # input is over, and this func doesn't consider that an error
 474             return
 475 
 476         comment = False
 477 
 478         for i, b in enumerate(chunk):
 479             # skip space, tab, line-feed, carriage-return, or form-feed
 480             if b in (9, 10, 11, 13, 32):
 481                 continue
 482 
 483             if b == 47 or b == 35: # ord('/'), ord('#')
 484                 read(r, i)
 485                 demand_comment(r)
 486                 comment = True
 487                 break
 488 
 489             # found start of next token
 490             read(r, i)
 491             return
 492 
 493         if not comment:
 494             read(r, len(chunk))
 495 
 496 
 497 def skip_line(r) -> None:
 498     while True:
 499         chunk = r.peek(1024)
 500         if len(chunk) == 0:
 501             return
 502 
 503         i = chunk.find(b'\n')
 504         if i >= 0:
 505             read(r, i + 1)
 506             return
 507 
 508         read(r, len(chunk))
 509 
 510 
 511 def skip_general_comment(r) -> None:
 512     while True:
 513         chunk = r.peek(1024)
 514         if len(chunk) == 0:
 515             raise ValueError(f'input data ended before an expected */')
 516 
 517         i = chunk.find(b'*')
 518         if i < 0:
 519             # no */ in this chunk, so skip it and try with the next one
 520             read(r, len(chunk))
 521             continue
 522 
 523         # skip right past the * just found, then check if a / follows it
 524         read(r, i + 1)
 525         if peek_byte(r) == 47: # ord('/')
 526             # got */, the end of this comment
 527             skip_byte(r)
 528             return
 529 
 530 
 531 def find_digits_end_index(chunk: bytes) -> int:
 532     i = 0
 533     for b in chunk:
 534         if 48 <= b <= 57:
 535             i += 1
 536         else:
 537             return i
 538 
 539     # all bytes (if any) were digits, so no end was found
 540     return -1
 541 
 542 
 543 def demand(r, what: bytes) -> None:
 544     lead = read(r, len(what))
 545     if not lead.startswith(what):
 546         lead = str(lead, encoding='utf-8')
 547         what = str(what, encoding='utf-8')
 548         raise ValueError(f'expected {what}, but got {lead} instead')
 549 
 550 
 551 def demand_comment(r) -> None:
 552     b = peek_byte(r)
 553     if b < 0:
 554         raise ValueError('unexpected end of input data')
 555     if b == 35: # ord('#')
 556         # handle single-line comment
 557         skip_line(r)
 558         return
 559 
 560     demand(r, b'/')
 561     b = peek_byte(r)
 562     if b < 0:
 563         raise ValueError('unexpected end of input data')
 564 
 565     if b == 47: # ord('/')
 566         # handle single-line comment
 567         skip_line(r)
 568         return
 569 
 570     if b == 42: # ord('*')
 571         # handle (potentially) multi-line comment
 572         skip_general_comment(r)
 573         return
 574 
 575     raise ValueError('expected * or another /, after a /')
 576 
 577 
 578 def json0(w, src, end) -> None:
 579     r = BufferedReader(src)
 580 
 581     # skip leading UTF-8 BOM (byte-order mark)
 582     if r.peek(3) == b'\xef\xbb\xbf':
 583         read(r, 3)
 584 
 585     # skip leading whitespace/comments
 586     seek_next_token(r)
 587 
 588     # emit a single output line, ending with a line-feed
 589     b = peek_byte(r)
 590     if b >= 0:
 591         handlers[b](w, r)
 592     else:
 593         # w.write(b'null')
 594         # treat empty(ish) input as invalid JSON
 595         raise ValueError('can\'t turn empty(ish) input into JSON')
 596 
 597     # deliberately run post-processing before checking for trailing-data
 598     # errors: for example, if post-proc func emits new line, errors will
 599     # show up on their separate line, which is nicer
 600     end(w)
 601 
 602     # ignore trailing whitespace/comment bytes, if present
 603     seek_next_token(r)
 604 
 605     # ignore trailing semicolon, if present
 606     b = peek_byte(r)
 607     if b == 59: # ord(';')
 608         read(r, 1)
 609         # ignore trailing whitespace/comment bytes, if present
 610         seek_next_token(r)
 611 
 612     if len(r.peek(1)) > 0:
 613         raise ValueError('unexpected trailing bytes in JSON data')
 614 
 615 
 616 def seems_url(s: str) -> bool:
 617     protocols = ('https://', 'http://', 'file://', 'ftp://', 'data:')
 618     return any(s.startswith(p) for p in protocols)
 619 
 620 
 621 def handle_json(w, r) -> None:
 622     def end(w) -> None:
 623         w.write(b'\n')
 624         w.flush()
 625     json0(w, r, end)
 626 
 627 
 628 def handle_json_lines(w, r) -> None:
 629     global pos, linenum
 630 
 631     items = 0
 632     linenum = 0
 633     w.write(b'[')
 634 
 635     while True:
 636         line = r.readline().lstrip()
 637         if not line:
 638             break
 639 
 640         pos = 1
 641         linenum += 1
 642 
 643         stripped = line.strip()
 644         if not stripped or stripped.startswith(b'//'):
 645             continue
 646 
 647         items += 1
 648         if items > 1:
 649             w.write(b',')
 650 
 651         json0(w, BytesIO(line), lambda w: w.flush())
 652 
 653     w.write(b']\n')
 654 
 655 
 656 start_args = 1
 657 handle_input = handle_json
 658 if len(argv) > 1 and argv[1] in ('-jl', '--jl', '-jsonl', '--jsonl'):
 659     start_args = 2
 660     handle_input = handle_json_lines
 661 
 662 if len(argv) > start_args:
 663     s = argv[start_args]
 664     if s != '-' and s != '--' and s.startswith('-'):
 665         print(f'json0: unsupported option {s}')
 666         exit(1)
 667 
 668 if len(argv) > start_args and argv[start_args] == '--':
 669     start_args += 1
 670 
 671 if len(argv) - 1 > start_args:
 672     print(f'json0: multiple inputs not allowed', file=stderr)
 673     exit(1)
 674 
 675 w = stdout.buffer
 676 name = argv[start_args] if len(argv) > start_args else '-'
 677 
 678 # values keeping track of the input-position, shown in case of errors
 679 pos = 1
 680 linenum = 1
 681 
 682 try:
 683     if name == '-':
 684         handle_input(w, stdin.buffer)
 685     elif seems_url(name):
 686         from urllib.request import urlopen
 687         with urlopen(name) as inp:
 688             handle_input(w, inp)
 689     else:
 690         with open(name, mode='rb') as inp:
 691             handle_input(w, inp)
 692 except BrokenPipeError:
 693     # quit quietly, instead of showing a confusing error message
 694     stderr.close()
 695     exit(0)
 696 except KeyboardInterrupt:
 697     exit(2)
 698 except Exception as e:
 699     stdout.write('\n')
 700     print(f'line {linenum}, pos {pos} : {e}', file=stderr)
 701     exit(1)