Aiden Grossman c745c54970
[LLVM] Prefer octal to hex for printf (#157884)
Hex escapes of the form \xff are not universally supported in the printf
implementations on the platforms that LLVM runs on (although they
apparently are in the shell builtins). Octal escapes are required to be
supported by POSIX. This patch converts all hex escapes to octal escapes
for compatibility reasons.

This came up when trying to turn on lit's internal shell by default for
llvm/. We started using /usr/bin/printf instead of the shell builtin on
MacOS, which does not support hex escapes.

I used the following python script to automate most of the conversion
with a few manual touchups needed:
```py
import sys

def process_line(to_process: str):
    output = ""
    i = 0
    while i < len(to_process):
        if to_process[i:i+2] == '\\x':
            hex_string = to_process[i+2:i+4]
            number = int(hex_string, 16)
            output += "\\"
            octal_string = oct(number)[2:]
            if len(octal_string) == 1:
                octal_string = "00" + octal_string
            elif len(octal_string) == 2:
                octal_string = "0" + octal_string
            assert(len(octal_string) == 3)
            output += octal_string
            i += 4
        else:
            output += to_process[i]
            i += 1
    return output

with open(sys.argv[1]) as input_file:
    lines = input_file.readlines()

for i, _ in enumerate(lines):
    lines[i] = process_line(lines[i])

with open(sys.argv[1], 'w') as output_file:
    output_file.writelines(lines)
```
2025-09-10 12:45:33 -07:00

42 lines
1.7 KiB
Plaintext

# Test various error cases
# Synthesize a header only cgdata.
# struct Header {
# uint64_t Magic;
# uint32_t Version;
# uint32_t DataKind;
# uint64_t OutlinedHashTreeOffset;
# uint64_t StableFunctionMapOffset;
# }
RUN: touch %t_empty.cgdata
RUN: not llvm-cgdata --show %t_empty.cgdata 2>&1 | FileCheck %s --check-prefix=EMPTY
EMPTY: {{.}}cgdata: empty codegen data
# Not a magic.
RUN: printf '\377' > %t_malformed.cgdata
RUN: not llvm-cgdata --show %t_malformed.cgdata 2>&1 | FileCheck %s --check-prefix=MALFORMED
MALFORMED: {{.}}cgdata: malformed codegen data
# The minimum header size is 24.
RUN: printf '\377cgdata\201' > %t_corrupt.cgdata
RUN: not llvm-cgdata --show %t_corrupt.cgdata 2>&1 | FileCheck %s --check-prefix=CORRUPT
CORRUPT: {{.}}cgdata: invalid codegen data (file header is corrupt)
# The current version 4 while the header says 5.
RUN: printf '\377cgdata\201' > %t_version.cgdata
RUN: printf '\005\000\000\000' >> %t_version.cgdata
RUN: printf '\000\000\000\000' >> %t_version.cgdata
RUN: printf '\040\000\000\000\000\000\000\000' >> %t_version.cgdata
RUN: printf '\040\000\000\000\000\000\000\000' >> %t_version.cgdata
RUN: not llvm-cgdata --show %t_version.cgdata 2>&1 | FileCheck %s --check-prefix=BAD_VERSION
BAD_VERSION: {{.}}cgdata: unsupported codegen data version
# Header says an outlined hash tree, but the file ends after the header.
RUN: printf '\377cgdata\201' > %t_eof.cgdata
RUN: printf '\002\000\000\000' >> %t_eof.cgdata
RUN: printf '\001\000\000\000' >> %t_eof.cgdata
RUN: printf '\040\000\000\000\000\000\000\000' >> %t_eof.cgdata
RUN: printf '\040\000\000\000\000\000\000\000' >> %t_eof.cgdata
RUN: not llvm-cgdata --show %t_eof.cgdata 2>&1 | FileCheck %s --check-prefix=EOF
EOF: {{.}}cgdata: end of File