#!/usr/bin/env python3 # Example usage: # $ cat /mnt/p4/b/zd/4changif/4chan_gif_2025_11.tar | ~/Software/ai-made/tar_index_stdin.py 2>/dev/null | head -n4000 | grep "0$" | sed "s/\s0$//g" | grep -v "0$" | xargs -d "\n" sh -c 'for args do echo $args; len="$(echo $args | sed "s/.*\s//g")"; off="$(echo $args | perl -pE "s/\s\d+$//g" | sed "s/.*\s//g")"; echo $off; tail -c+$(expr $off + 1) /mnt/p4/b/zd/4changif/4chan_gif_2025_11.tar | head -c $len > /tmp/a; mpv --loop /tmp/a; done' _ import sys import struct BLOCK = 512 def parse_octal(b: bytes) -> int: b = b.strip(b"\x00 ").strip() if not b: return 0 return int(b, 8) def read_exact(f, n): data = f.read(n) if len(data) != n: raise EOFError(f"Unexpected EOF while reading {n} bytes") return data def main(): # We'll read from stdin as a binary stream. f = sys.stdin.buffer idx = 0 stream_offset = 0 # offset in bytes from start of tar stream # Print: path, offset, length, typeflag sys.stdout.write(f"path\toffset\tlength\ttypeflag\n") # tar can end with two all-zero blocks while True: header = f.read(BLOCK) if len(header) == 0: break # clean EOF if len(header) < BLOCK: raise EOFError("Unexpected EOF in header block") if header == b"\x00" * BLOCK: # Could be end-of-archive marker; consume possible second zero block # but allow for both. nxt = f.read(BLOCK) if nxt == b"": break if nxt != b"\x00" * BLOCK: # Not expected in well-formed tar, but we can treat it as error. raise ValueError("Invalid tar: zero block terminator followed by nonzero block") break # POSIX ustar header fields: # name[100], mode[8], uid[8], gid[8], size[12], mtime[12], # chksum[8], typeflag[1], linkname[100], magic[6], version[2], # uname[32], gname[32], devmajor[8], devminor[8], prefix[155] name = header[0:100] prefix = header[345:500] # prefix is 155 bytes: 346..500 (approx; inclusive) size_field = header[124:136] # 12 bytes typeflag = header[156:157] # 1 byte name_str = name.split(b"\x00", 1)[0].decode('utf-8', errors='replace') prefix_str = prefix.split(b"\x00", 1)[0].decode('utf-8', errors='replace') if prefix_str: full_name = prefix_str + "/" + name_str else: full_name = name_str size = parse_octal(size_field) # Data payload starts right after header data_offset = stream_offset + BLOCK data_length = size # For tar, files are stored in 512-byte blocks; if size isn't multiple of 512, # the remainder is padding included in the archive stream. data_block_len = ((size + BLOCK - 1) // BLOCK) * BLOCK # Emit index line. # We'll include typeflag so you can distinguish regular files vs symlinks/etc. tf = typeflag.decode('ascii', errors='replace') if not full_name: full_name = f"" # Print: pathoffsetlengthtypeflag sys.stdout.write(f"{full_name}\t{data_offset}\t{data_length}\t{tf}\n") # Skip over payload blocks to next header if data_block_len: # We can seek within stdin only if it's seekable; assume not. # So we must read and discard. _ = read_exact(f, data_block_len) stream_offset += BLOCK + data_block_len idx += 1 if __name__ == "__main__": main()