feat(02-01): implement INF parser with encoding detection and token resolution
- ParsedInf dataclass: driver_names, inf_filename, architecture, has_cat_file, unused_files - _detect_encoding(): BOM sniffing for UTF-16 LE/BE, UTF-8 BOM, cp1252 fallback - _resolve_tokens(): regex %TOKEN% expansion from [Strings] section dict - parse_inf(): RawConfigParser(strict=False) with optionxform=str to preserve case - Architecture detection: NTamd64->x64, NTarm64->arm64, undecorated->x86, mixed->None - Deduplication via set(); sorted() for deterministic dropdown order - [Rule 1 - Bug] Fixed configparser key lowercasing by setting optionxform=str - All 16 tests pass, zero regressions in 40-test suite
This commit is contained in:
@@ -0,0 +1,173 @@
|
|||||||
|
"""
|
||||||
|
INF parser service for Windows driver INF files.
|
||||||
|
|
||||||
|
Extracts driver names (DriverDesc), resolves %TOKEN% references,
|
||||||
|
auto-detects encoding (ANSI/UTF-8/UTF-16), handles multi-model INFs,
|
||||||
|
detects unused files, architecture, and presence of .cat files.
|
||||||
|
|
||||||
|
Source: Microsoft WDK — General Syntax Rules for INF Files
|
||||||
|
https://learn.microsoft.com/en-us/windows-hardware/drivers/install/general-syntax-rules-for-inf-files
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import configparser
|
||||||
|
import re
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class ParsedInf:
|
||||||
|
"""Result of parsing a Windows INF file."""
|
||||||
|
|
||||||
|
driver_names: list[str] # resolved DriverDesc values, deduplicated and sorted
|
||||||
|
inf_filename: str # which .inf file was parsed (basename from ZIP)
|
||||||
|
architecture: str | None # 'x64', 'x86', 'arm64', or None if ambiguous/unknown
|
||||||
|
has_cat_file: bool # True if a .cat file exists in the ZIP member list
|
||||||
|
unused_files: list[str] # ZIP members not referenced anywhere in the INF text
|
||||||
|
|
||||||
|
|
||||||
|
def _detect_encoding(raw: bytes) -> str:
|
||||||
|
"""Detect INF file encoding by sniffing BOM bytes.
|
||||||
|
|
||||||
|
INF files from real vendors arrive as:
|
||||||
|
- ANSI / Windows-1252 (cp1252) — most legacy drivers
|
||||||
|
- UTF-8 with BOM — modern drivers
|
||||||
|
- UTF-16 LE with BOM — HP/Canon x64 signed drivers (most common UTF-16)
|
||||||
|
- UTF-16 BE with BOM — rare
|
||||||
|
|
||||||
|
Source: Microsoft WDK — general-syntax-rules-for-inf-files
|
||||||
|
"""
|
||||||
|
if raw[:2] in (b"\xff\xfe", b"\xfe\xff"):
|
||||||
|
return "utf-16" # UTF-16 LE or BE with BOM
|
||||||
|
if raw[:3] == b"\xef\xbb\xbf":
|
||||||
|
return "utf-8-sig" # UTF-8 with BOM
|
||||||
|
return "cp1252" # ANSI / Windows-1252 safe fallback
|
||||||
|
|
||||||
|
|
||||||
|
def _resolve_tokens(value: str, strings: dict[str, str]) -> str:
|
||||||
|
"""Expand %TOKEN% placeholders using the [Strings] section lookup dict.
|
||||||
|
|
||||||
|
Keys in ``strings`` must already be lowercased (configparser lowercases
|
||||||
|
option names by default).
|
||||||
|
|
||||||
|
Source: Microsoft WDK — general-syntax-rules-for-inf-files — strkey% syntax
|
||||||
|
"""
|
||||||
|
|
||||||
|
def replacer(match: re.Match) -> str:
|
||||||
|
key = match.group(1).lower()
|
||||||
|
return strings.get(key, match.group(0))
|
||||||
|
|
||||||
|
return re.sub(r"%([^%]+)%", replacer, value)
|
||||||
|
|
||||||
|
|
||||||
|
def parse_inf(inf_text: str, inf_filename: str, zip_names: list[str]) -> ParsedInf:
|
||||||
|
"""Parse a Windows INF file and extract driver metadata.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
inf_text: Decoded text content of the .inf file.
|
||||||
|
inf_filename: Name of the .inf file (used as-is in the result).
|
||||||
|
zip_names: List of all member paths from the containing ZIP archive.
|
||||||
|
Used for unused-file detection and .cat presence check.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
ParsedInf dataclass with driver_names, architecture, has_cat_file,
|
||||||
|
unused_files, and inf_filename.
|
||||||
|
|
||||||
|
Notes:
|
||||||
|
- Uses RawConfigParser (NOT ConfigParser) to avoid %(interpolation)s
|
||||||
|
interference with %TOKEN% INF syntax.
|
||||||
|
- strict=False is required because real INF files frequently have
|
||||||
|
duplicate option keys within a section (multiple hardware IDs).
|
||||||
|
- Architecture detection: if exactly one arch hint found -> return it;
|
||||||
|
multiple arch hints -> None (ambiguous / multi-arch INF).
|
||||||
|
- driver_names are sorted for deterministic dropdown order.
|
||||||
|
"""
|
||||||
|
parser = configparser.RawConfigParser(
|
||||||
|
comment_prefixes=(";", "#"),
|
||||||
|
strict=False, # real INFs have duplicate keys
|
||||||
|
delimiters=("=",),
|
||||||
|
)
|
||||||
|
# Preserve original option-key casing so DriverDesc literals keep their case.
|
||||||
|
# configparser lowercases keys by default, which would mangle "Acme SuperPrint 9000"
|
||||||
|
# into "acme superprint 9000". We disable that behaviour here and manually lowercase
|
||||||
|
# only when building the [Strings] lookup dict.
|
||||||
|
parser.optionxform = str # type: ignore[assignment]
|
||||||
|
parser.read_string(inf_text)
|
||||||
|
|
||||||
|
# Build strings lookup with LOWERCASED keys (case-insensitive token resolution).
|
||||||
|
# INF token references are case-insensitive per WDK spec.
|
||||||
|
strings: dict[str, str] = {}
|
||||||
|
if parser.has_section("Strings"):
|
||||||
|
for key, val in parser.items("Strings"):
|
||||||
|
# INF string values are typically surrounded by double-quotes; strip them.
|
||||||
|
strings[key.lower()] = val.strip('"')
|
||||||
|
|
||||||
|
# Collect Models section base names from [Manufacturer]
|
||||||
|
# Format per WDK: mfg-id = models-section-name[,target-OS-version[,target-OS-version...]]
|
||||||
|
models_section_names: list[str] = []
|
||||||
|
if parser.has_section("Manufacturer"):
|
||||||
|
for _mfg_key, mfg_val in parser.items("Manufacturer"):
|
||||||
|
# Resolve any %TOKEN% in the manufacturer value (rare, but safe)
|
||||||
|
resolved_val = _resolve_tokens(mfg_val, strings)
|
||||||
|
parts = [p.strip() for p in resolved_val.split(",")]
|
||||||
|
if parts:
|
||||||
|
models_section_names.append(parts[0])
|
||||||
|
|
||||||
|
# For each referenced Models section base name, find all matching sections
|
||||||
|
# (undecorated, .NTamd64, .NTarm64, .NTx86, etc.) and extract DriverDesc entries.
|
||||||
|
driver_names: set[str] = set()
|
||||||
|
arch_hints: set[str] = set()
|
||||||
|
|
||||||
|
all_sections_lower = {s.lower(): s for s in parser.sections()}
|
||||||
|
|
||||||
|
for base_name in models_section_names:
|
||||||
|
base_lower = base_name.lower()
|
||||||
|
for section_lower, section in all_sections_lower.items():
|
||||||
|
# Match: exact base name (undecorated) OR base name + .NT<arch> decoration
|
||||||
|
if section_lower == base_lower:
|
||||||
|
# Undecorated section — architecture hint: x86
|
||||||
|
arch_hints.add("x86")
|
||||||
|
suffix = ""
|
||||||
|
elif section_lower.startswith(base_lower + ".nt"):
|
||||||
|
suffix = section_lower[len(base_lower):] # e.g. ".ntamd64"
|
||||||
|
if "amd64" in suffix:
|
||||||
|
arch_hints.add("x64")
|
||||||
|
elif "arm64" in suffix:
|
||||||
|
arch_hints.add("arm64")
|
||||||
|
elif "x86" in suffix:
|
||||||
|
arch_hints.add("x86")
|
||||||
|
else:
|
||||||
|
# Generic .NT decoration (no specific arch) — treat as x86
|
||||||
|
arch_hints.add("x86")
|
||||||
|
else:
|
||||||
|
continue
|
||||||
|
|
||||||
|
# Each option key in a Models section is a device-description (DriverDesc)
|
||||||
|
for key, _val in parser.items(section):
|
||||||
|
resolved = _resolve_tokens(key, strings)
|
||||||
|
# Skip empty, purely numeric, or clearly non-driver-name entries
|
||||||
|
if resolved and not resolved.isdigit():
|
||||||
|
driver_names.add(resolved)
|
||||||
|
|
||||||
|
# Architecture: unambiguous only when exactly one arch hint found
|
||||||
|
architecture: str | None = arch_hints.pop() if len(arch_hints) == 1 else None
|
||||||
|
|
||||||
|
# .cat file detection
|
||||||
|
has_cat_file = any(name.lower().endswith(".cat") for name in zip_names)
|
||||||
|
|
||||||
|
# Unused files: ZIP members whose basename does not appear anywhere in INF text
|
||||||
|
inf_lower = inf_text.lower()
|
||||||
|
unused_files: list[str] = []
|
||||||
|
for member in zip_names:
|
||||||
|
# Normalise path separators, then take basename
|
||||||
|
basename = member.replace("\\", "/").rsplit("/", 1)[-1]
|
||||||
|
if basename.lower() not in inf_lower:
|
||||||
|
unused_files.append(member)
|
||||||
|
|
||||||
|
return ParsedInf(
|
||||||
|
driver_names=sorted(driver_names),
|
||||||
|
inf_filename=inf_filename,
|
||||||
|
architecture=architecture,
|
||||||
|
has_cat_file=has_cat_file,
|
||||||
|
unused_files=unused_files,
|
||||||
|
)
|
||||||
Reference in New Issue
Block a user