File: si.py
   1 #!/usr/bin/python
   2 
   3 # The MIT License (MIT)
   4 #
   5 # Copyright (c) 2026 pacman64
   6 #
   7 # Permission is hereby granted, free of charge, to any person obtaining a copy
   8 # of this software and associated documentation files (the "Software"), to deal
   9 # in the Software without restriction, including without limitation the rights
  10 # to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
  11 # copies of the Software, and to permit persons to whom the Software is
  12 # furnished to do so, subject to the following conditions:
  13 #
  14 # The above copyright notice and this permission notice shall be included in
  15 # all copies or substantial portions of the Software.
  16 #
  17 # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
  18 # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
  19 # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
  20 # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
  21 # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
  22 # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
  23 # SOFTWARE.
  24 
  25 
  26 from base64 import b64decode, b64encode
  27 from io import BytesIO
  28 from re import compile as compile_re, Pattern
  29 from socket import socket
  30 from sys import argv, exit, stderr, stdin
  31 from typing import Callable, Dict, List, Tuple
  32 from webbrowser import open_new_tab
  33 
  34 
  35 info = '''
  36 si [options...]
  37 
  38 
  39 Show It shows data read from standard-input, using your default web browser
  40 by auto-opening tabs, auto-detecing the data-format, and using a random port
  41 among those available.
  42 
  43 The localhost connection is available only until all data are transferred:
  44 this means refreshing your browser tab will lose your content, replacing it
  45 with a server-not-found message page.
  46 
  47 Dozens of common data-formats are recognized when piped from stdin, such as
  48 
  49   - HTML (web pages)
  50   - PDF
  51   - pictures (PNG, JPEG, SVG, WEBP, HEIC, AVIF, GIF, BMP)
  52   - audio (AAC, MP3, FLAC, WAV, AU, MIDI)
  53   - video (MP4, MOV, WEBM, MKV, AVI)
  54   - JSON
  55   - generic UTF-8 plain-text
  56 
  57 All options can start with either a single or a double leading dash:
  58 
  59     -h, -help    show this help message
  60 '''
  61 
  62 # handle standard help cmd-line options, quitting right away in that case
  63 if len(argv) > 1 and argv[1] in ('-h', '--h', '-help', '--help'):
  64     print(info.strip())
  65     exit(0)
  66 
  67 if len(argv) > 1:
  68     s = argv[1]
  69     if s != '-' and s != '--' and s.startswith('-'):
  70         print(f'si: unsupported option {s}', file=stderr)
  71         exit(1)
  72 
  73 
  74 # hdr_dispatch groups format-description-groups by their first byte, thus
  75 # shortening total lookups for some data header: notice how the `ftyp` data
  76 # formats aren't handled here, since these can include any byte in parts of
  77 # their first few bytes
  78 hdr_dispatch: Dict[int, List[Tuple[bytes, str]]] = {
  79     0x00: [
  80         (b'\x00\x00\x01\xba', 'video/mpeg'),
  81         (b'\x00\x00\x01\xb3', 'video/mpeg'),
  82         (b'\x00\x00\x01\x00', 'image/x-icon'),
  83         (b'\x00\x00\x02\x00', 'image/vnd.microsoft.icon'), # .cur files
  84         (b'\x00asm', 'application/wasm'),
  85     ],
  86     0x1a: [(b'\x1a\x45\xdf\xa3', 'video/webm')], # matches general MKV format
  87     0x1f: [(b'\x1f\x8b\x08', 'application/gzip')],
  88     0x23: [
  89         (b'#! ', 'text/plain; charset=UTF-8'),
  90         (b'#!/', 'text/plain; charset=UTF-8'),
  91     ],
  92     0x25: [(b'%PDF', 'application/pdf'), (b'%!PS', 'application/postscript')],
  93     0x28: [(b'\x28\xb5\x2f\xfd', 'application/zstd')],
  94     0x2e: [(b'.snd', 'audio/basic')],
  95     0x47: [(b'GIF87a', 'image/gif'), (b'GIF89a', 'image/gif')],
  96     0x49: [
  97         # some MP3s start with an ID3 meta-data section
  98         (b'ID3\x02', 'audio/mpeg'),
  99         (b'ID3\x03', 'audio/mpeg'),
 100         (b'ID3\x04', 'audio/mpeg'),
 101         (b'II*\x00', 'image/tiff'),
 102     ],
 103     0x4d: [(b'MM\x00*', 'image/tiff'), (b'MThd', 'audio/midi')],
 104     0x4f: [(b'OggS', 'audio/ogg')],
 105     0x50: [(b'PK\x03\x04', 'application/zip')],
 106     0x53: [(b'SQLite format 3\x00', 'application/x-sqlite3')],
 107     0x63: [(b'caff\x00\x01\x00\x00', 'audio/x-caf')],
 108     0x66: [(b'fLaC', 'audio/x-flac')],
 109     0x7b: [(b'{\\rtf', 'application/rtf')],
 110     0x7f: [(b'\x7fELF', 'application/x-elf')],
 111     0x89: [(b'\x89PNG\x0d\x0a\x1a\x0a', 'image/png')],
 112     0xff: [
 113         (b'\xff\xd8\xff', 'image/jpeg'),
 114         # handle common ways MP3 data start
 115         (b'\xff\xf3\x48\xc4\x00', 'audio/mpeg'),
 116         (b'\xff\xfb', 'audio/mpeg'),
 117     ],
 118 }
 119 
 120 
 121 # ftyp_types helps func match_ftyp auto-detect MPEG-4-like formats
 122 ftyp_types: Tuple[Tuple[bytes, str]] = (
 123     (b'M4A ', 'audio/aac'),
 124     (b'M4A\x00', 'audio/aac'),
 125     (b'mp42', 'video/x-m4v'),
 126     (b'dash', 'audio/aac'),
 127     (b'isom', 'video/mp4'),
 128     # (b'isom', 'audio/aac'),
 129     (b'MSNV', 'video/mp4'),
 130     (b'qt  ', 'video/quicktime'),
 131     (b'heic', 'image/heic'),
 132     (b'avif', 'image/avif'),
 133 )
 134 
 135 # xmlish_heuristics helps func guess_mime auto-detect HTML, SVG, and XML
 136 xmlish_heuristics: Tuple[Tuple[bytes, str]] = (
 137     (b'<html>', 'text/html'), (b'<html ', 'text/html'),
 138     (b'<head>', 'text/html'), (b'<head ', 'text/html'),
 139     (b'<body>', 'text/html'), (b'<body ', 'text/html'),
 140     (b'<!DOCTYPE html>', 'text/html'), (b'<!DOCTYPE html ', 'text/html'),
 141     (b'<svg>', 'image/svg+xml'), (b'<svg ', 'image/svg+xml'),
 142     (b'<?xml>', 'application/xml'), (b'<?xml ', 'application/xml'),
 143 )
 144 
 145 # json_heuristics helps func guess_mime auto-detect JSON via regexes:
 146 # it's not perfect, but it seems effective-enough in practice
 147 json_heuristics: Tuple[Pattern] = (
 148     compile_re(b'''^\\s*\\{\\s*"'''),
 149     compile_re(b'''^\\s*\\{\\s*\\['''),
 150     compile_re(b'''^\\s*\\[\\s*"'''),
 151     compile_re(b'''^\\s*\\[\\s*\\{'''),
 152     compile_re(b'''^\\s*\\[\\s*\\['''),
 153 )
 154 
 155 
 156 def exact_match(header: bytes, maybe: bytes) -> bool:
 157     enough_bytes = len(header) >= len(maybe)
 158     return enough_bytes and all(x == y for x, y in zip(header, maybe))
 159 
 160 
 161 def match_riff(header: bytes) -> str:
 162     if len(header) < 12 or not header.startswith(b'RIFF'):
 163         return ''
 164 
 165     if header.find(b'WEBP', 8, 12) == 8:
 166         return 'image/webp'
 167     if header.find(b'WAVE', 8, 12) == 8:
 168         return 'audio/x-wav'
 169     if header.find(b'AVI ', 8, 12) == 8:
 170         return 'video/avi'
 171     return ''
 172 
 173 
 174 def match_form(header: bytes) -> str:
 175     if len(header) < 12 or not header.startswith(b'FORM'):
 176         return ''
 177 
 178     if header.find(b'AIFF', 8, 12) == 8:
 179         return 'audio/aiff'
 180     if header.find(b'AIFC', 8, 12) == 8:
 181         return 'audio/aiff'
 182     return ''
 183 
 184 
 185 def match_ftyp(header: bytes) -> str:
 186     # first 4 bytes can be anything, next 4 bytes must be ASCII 'ftyp'
 187     if len(header) < 12 or header.find(b'ftyp', 4, 8) != 4:
 188         return ''
 189 
 190     # next 4 bytes after the ASCII 'ftyp' declare the data-format
 191     for marker, mime in ftyp_types:
 192         if header.find(marker, 8, 12) == 8:
 193             return mime
 194 
 195     return ''
 196 
 197 
 198 def guess_mime(header: bytes, fallback: str) -> str:
 199     # no bytes, no match
 200     if len(header) == 0:
 201         return fallback
 202 
 203     # check the MPEG-4-like formats, the RIFF formats, and AIFF audio
 204     for f in (match_ftyp, match_riff, match_form):
 205         m = f(header)
 206         if m != '':
 207             return m
 208 
 209     # maybe it's a bitmap picture, which almost always has 40 on 15th byte
 210     if header.startswith(b'BM') and header.find(b'\x28', 8, 16) == 14:
 211         return 'image/x-bmp'
 212 
 213     # check general lookup-table
 214     if header[0] in hdr_dispatch:
 215         for maybe in hdr_dispatch[header[0]]:
 216             if exact_match(header, maybe[0]):
 217                 return maybe[1]
 218 
 219     # try HTML, SVG, and even generic XML
 220     if header.find(b'<', 0, 8) >= 0:
 221         for marker, mime in xmlish_heuristics:
 222             if header.find(marker, 0, 64) >= 0:
 223                 return mime
 224 
 225     # try some common cases for JSON
 226     for pattern in json_heuristics:
 227         if pattern.match(header):
 228             return 'application/json'
 229 
 230     # nothing matched
 231     return fallback
 232 
 233 
 234 def show_it(conn, start: bytes, rest) -> None:
 235     # handle base64-encoded data-URIs
 236     if start.startswith(b'data:'):
 237         i = start.find(b';base64,', 0, 64)
 238         if i > 0:
 239             mime_type = str(start[len('data:'):i], encoding='utf-8')
 240             encoded = BytesIO()
 241             encoded.write(start[i + len(';base64,'):])
 242             encoded.write(rest.read())
 243             decoded = b64decode(encoded.getvalue())
 244             encoded.close()
 245 
 246             inp = BytesIO(decoded)
 247             if mime_type == '':
 248                 start = inp.read(4096)
 249                 mime_type = guess_mime(start, 'text/plain; charset=UTF-8')
 250                 show_it_as(conn, start, inp, mime_type)
 251             else:
 252                 show_it_as(conn, bytes(), inp, mime_type)
 253             return
 254 
 255     mime_type = guess_mime(start, 'text/plain; charset=UTF-8')
 256     return show_it_as(conn, start, rest, mime_type)
 257 
 258 
 259 def show_it_as(conn, start: bytes, rest, mime_type: str) -> None:
 260     # read-ignore all client headers
 261     while True:
 262         if conn.recv(1024).endswith(b'\r\n\r\n'):
 263             break
 264 
 265     # web-browsers insist on auto-downloads when given wave or flac audio
 266     for e in ('audio/x-wav', 'audio/x-flac'):
 267         if e == mime_type:
 268             handle_sound_workaround(conn, mime_type, start, rest)
 269             return
 270 
 271     # web-browsers insist on auto-downloads when given bitmap pictures
 272     if mime_type == 'image/x-bmp':
 273         handle_image_workaround(conn, mime_type, start, rest)
 274         return
 275 
 276     conn.sendall(b'HTTP/1.1 200 OK\r\n')
 277     conn.sendall(bytes(f'Content-Type: {mime_type}\r\n', encoding='utf-8'))
 278     conn.sendall(b'Content-Disposition: inline\r\nConnection: close\r\n\r\n')
 279     conn.sendall(start)
 280     conn.sendfile(rest)
 281 
 282 
 283 def chunked_write(dest, src) -> None:
 284     while True:
 285         chunk = src.read(32 * 1024)
 286         if not chunk:
 287             return
 288         dest.write(chunk)
 289 
 290 
 291 def handle_sound_workaround(conn, mime: str, start: bytes, rest) -> None:
 292     def emit_inner_body() -> None:
 293         data = BytesIO()
 294         s = f'    <audio controls autofocus src="data:{mime};base64,'
 295         conn.sendall(bytes(s, encoding='utf-8'))
 296         data.write(start)
 297         chunked_write(data, rest)
 298         conn.sendall(b64encode(data.getvalue()))
 299         conn.sendall(b'"></audio>\n')
 300         data.close()
 301     handle_workaround(conn, 'Sound', emit_inner_body)
 302 
 303 
 304 def handle_image_workaround(conn, mime: str, start: bytes, rest) -> None:
 305     def emit_inner_body() -> None:
 306         data = BytesIO()
 307         s = bytes(f'    <img src="data:{mime};base64,', encoding='utf-8')
 308         conn.sendall(s)
 309         data.write(start)
 310         chunked_write(data, rest)
 311         conn.sendall(b64encode(data.getvalue()))
 312         conn.sendall(b'">\n')
 313         data.close()
 314     handle_workaround(conn, 'Bitmap Picture', emit_inner_body)
 315 
 316 
 317 start = '''
 318 <!DOCTYPE html>
 319 <html lang="en">
 320 <head>
 321     <meta charset="UTF-8">
 322     <link rel="icon" href="data:,">
 323     <meta name="viewport" content="width=device-width, initial-scale=1.0">
 324     <title>%%</title>
 325     <style>
 326         body {
 327             margin: 0;
 328             padding: 0;
 329         }
 330 
 331         audio {
 332             display: block;
 333             margin: auto;
 334             width: 90vw;
 335         }
 336 
 337         img {
 338             display: block;
 339             margin: auto;
 340         }
 341     </style>
 342 </head>
 343 <body>
 344 '''
 345 
 346 def handle_workaround(conn, title: str, handle_inner_body: Callable) -> None:
 347     s = bytes(start.replace('%%', title).lstrip('\n'), encoding='utf-8')
 348     conn.sendall(b'HTTP/1.1 200 OK\r\n')
 349     conn.sendall(b'Content-Type: text/html; charset=UTF-8\r\n')
 350     conn.sendall(b'Content-Disposition: inline\r\nConnection: close\r\n\r\n')
 351     conn.sendall(s)
 352     handle_inner_body()
 353     conn.sendall(b'</body>\n</html>\n')
 354 
 355 
 356 try:
 357     # opening socket on port 0 randomly picks an available port
 358     sock = socket()
 359     sock.bind(('localhost', 0))
 360     port = sock.getsockname()[1]
 361     sock.settimeout(10.0)
 362     # only handle one thing at a time, since it's a one-off server
 363     sock.listen(1)
 364 
 365     open_new_tab(f'http://localhost:{port}')
 366 
 367     # handle only a single request-response cycle
 368     conn, addr = sock.accept()
 369     show_it(conn, stdin.buffer.read(4096), stdin.buffer)
 370     conn.close()
 371 
 372     sock.close()
 373 except Exception as e:
 374     print(str(e), file=stderr)
 375     exit(1)