[llvm] [z/OS] Add z/OS archive reading support (PR #187110)
James Henderson via llvm-commits
llvm-commits at lists.llvm.org
Wed Apr 29 01:07:16 PDT 2026
================
@@ -0,0 +1,400 @@
+#!/usr/bin/env python3
+"""Generate z/OS archive files
+
+z/OS archives use EBCDIC encoding for headers, magic bytes, and symbol names.
+This script generates archives in place to avoid reliance on canned binaries.
+
+Usage examples:
+ # Valid archive with one member and symbol table:
+ %python %S/Inputs/generate_zos_archive.py --output %t.a \
+ --symtab "foo:0" --member foo.o:%S/Inputs/foo.o
+
+ # Empty archive:
+ %python %S/Inputs/generate_zos_archive.py --output %t.a --empty
+
+ # Malformed member header: bad terminator
+ %python %S/Inputs/generate_zos_archive.py --output %t.a \
+ --member foo.o --bad-terminator
+
+ # Malformed __.SYMDEF header: bad terminator
+ %python %S/Inputs/generate_zos_archive.py --output %t.a \
+ --member foo.o --symtab foo:0 --malform-symtab-hdr bad-terminator
+
+ # Member with explicit hex content:
+ %python %S/Inputs/generate_zos_archive.py --output %t.a \
+ --member foo.o:hex:deadbeef
+"""
+
+import argparse
+import struct
+import sys
+import os
+
+# EBCDIC / ASCII conversion table
+# fmt: off
+ASCII_TO_EBCDIC_TABLE = (
+ 0x00,0x01,0x02,0x03,0x37,0x2D,0x2E,0x2F,0x16,0x05,0x15,0x0B,0x0C,0x0D,0x0E,0x0F,
+ 0x10,0x11,0x12,0x13,0x3C,0x3D,0x32,0x26,0x18,0x19,0x3F,0x27,0x1C,0x1D,0x1E,0x1F,
+ 0x40,0x5A,0x7F,0x7B,0x5B,0x6C,0x50,0x7D,0x4D,0x5D,0x5C,0x4E,0x6B,0x60,0x4B,0x61,
+ 0xF0,0xF1,0xF2,0xF3,0xF4,0xF5,0xF6,0xF7,0xF8,0xF9,0x7A,0x5E,0x4C,0x7E,0x6E,0x6F,
+ 0x7C,0xC1,0xC2,0xC3,0xC4,0xC5,0xC6,0xC7,0xC8,0xC9,0xD1,0xD2,0xD3,0xD4,0xD5,0xD6,
+ 0xD7,0xD8,0xD9,0xE2,0xE3,0xE4,0xE5,0xE6,0xE7,0xE8,0xE9,0xAD,0xE0,0xBD,0x5F,0x6D,
+ 0x79,0x81,0x82,0x83,0x84,0x85,0x86,0x87,0x88,0x89,0x91,0x92,0x93,0x94,0x95,0x96,
+ 0x97,0x98,0x99,0xA2,0xA3,0xA4,0xA5,0xA6,0xA7,0xA8,0xA9,0xC0,0x4F,0xD0,0xA1,0x07,
+)
+# fmt: on
+
+
+def ascii_to_ebcdic(s):
+ """Convert an ASCII string/bytes to EBCDIC (IBM-1047)."""
+ if isinstance(s, str):
+ s = s.encode("ascii")
+ return bytes(ASCII_TO_EBCDIC_TABLE[b] for b in s)
+
+
+def ebcdic_pad(s, width, pad_char=" "):
+ """Convert ASCII string to EBCDIC, right-padded with EBCDIC spaces."""
+ ascii_padded = s.ljust(width, pad_char)
+ return ascii_to_ebcdic(ascii_padded)
+
+
+# z/OS archive magic: "!<arch>\n" in EBCDIC
+ZOS_MAGIC = b"\x5a\x4c\x81\x99\x83\x88\x6e\x15"
+
+# Terminator: "`\n" in EBCDIC
+ZOS_TERMINATOR = b"\x79\x15"
+
+# EBCDIC newline for padding
----------------
jh7370 wrote:
Nit: here and in various other places, need a trailing "." at the end of the comment.
https://github.com/llvm/llvm-project/pull/187110
More information about the llvm-commits
mailing list