Detect the encoding of a file, with special handling for UTF-16. Args: file_path: Path to the file to analyze Returns: String indicating the detected encoding
(file_path)
| 127 | return retval |
| 128 | |
| 129 | def detect_encoding(file_path): |
| 130 | """ |
| 131 | Detect the encoding of a file, with special handling for UTF-16. |
| 132 | |
| 133 | Args: |
| 134 | file_path: Path to the file to analyze |
| 135 | |
| 136 | Returns: |
| 137 | String indicating the detected encoding |
| 138 | """ |
| 139 | with open(file_path, 'rb') as f: |
| 140 | # Read the first few bytes to check for BOM |
| 141 | first_bytes = f.read(4) |
| 142 | # Check for UTF-16 BOM (Little Endian) |
| 143 | if first_bytes.startswith(b'\xff\xfe'): |
| 144 | return 'utf-16-le' |
| 145 | # Check for UTF-16 BOM (Big Endian) |
| 146 | elif first_bytes.startswith(b'\xfe\xff'): |
| 147 | return 'utf-16-be' |
| 148 | # Check for UTF-8 BOM |
| 149 | elif first_bytes.startswith(b'\xef\xbb\xbf'): |
| 150 | return 'utf-8-sig' |
| 151 | # Reset file pointer and use chardet for other encodings |
| 152 | f.seek(0) |
| 153 | raw_data = f.read(10000) # Read a chunk for detection |
| 154 | result = chardet.detect(raw_data) |
| 155 | return result['encoding'] |
| 156 | |
| 157 | def process_binary_file(file_path, output_file): |
| 158 | """ |
no outgoing calls
no test coverage detected