| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716717718719720721722723724725726727728729730731732733734735736737738739740741742743744745746747748749750751752753754755756757758759760761762763764765766767768769770771772773774775776777778779780781782783784785786787788789790791792793794795796797798799800801802803804805806807808809810811812813814815816817818819820821822823824825826827828829830831832833834835836837838839840841842843844845846847 |
- """
- Base validator with common validation logic for document files.
- """
- import re
- from pathlib import Path
- import defusedxml.minidom
- import lxml.etree
- class BaseSchemaValidator:
- IGNORED_VALIDATION_ERRORS = [
- "hyphenationZone",
- "purl.org/dc/terms",
- ]
- UNIQUE_ID_REQUIREMENTS = {
- "comment": ("id", "file"),
- "commentrangestart": ("id", "file"),
- "commentrangeend": ("id", "file"),
- "bookmarkstart": ("id", "file"),
- "bookmarkend": ("id", "file"),
- "sldid": ("id", "file"),
- "sldmasterid": ("id", "global"),
- "sldlayoutid": ("id", "global"),
- "cm": ("authorid", "file"),
- "sheet": ("sheetid", "file"),
- "definedname": ("id", "file"),
- "cxnsp": ("id", "file"),
- "sp": ("id", "file"),
- "pic": ("id", "file"),
- "grpsp": ("id", "file"),
- }
- EXCLUDED_ID_CONTAINERS = {
- "sectionlst",
- }
- ELEMENT_RELATIONSHIP_TYPES = {}
- SCHEMA_MAPPINGS = {
- "word": "ISO-IEC29500-4_2016/wml.xsd",
- "ppt": "ISO-IEC29500-4_2016/pml.xsd",
- "xl": "ISO-IEC29500-4_2016/sml.xsd",
- "[Content_Types].xml": "ecma/fouth-edition/opc-contentTypes.xsd",
- "app.xml": "ISO-IEC29500-4_2016/shared-documentPropertiesExtended.xsd",
- "core.xml": "ecma/fouth-edition/opc-coreProperties.xsd",
- "custom.xml": "ISO-IEC29500-4_2016/shared-documentPropertiesCustom.xsd",
- ".rels": "ecma/fouth-edition/opc-relationships.xsd",
- "people.xml": "microsoft/wml-2012.xsd",
- "commentsIds.xml": "microsoft/wml-cid-2016.xsd",
- "commentsExtensible.xml": "microsoft/wml-cex-2018.xsd",
- "commentsExtended.xml": "microsoft/wml-2012.xsd",
- "chart": "ISO-IEC29500-4_2016/dml-chart.xsd",
- "theme": "ISO-IEC29500-4_2016/dml-main.xsd",
- "drawing": "ISO-IEC29500-4_2016/dml-main.xsd",
- }
- MC_NAMESPACE = "http://schemas.openxmlformats.org/markup-compatibility/2006"
- XML_NAMESPACE = "http://www.w3.org/XML/1998/namespace"
- PACKAGE_RELATIONSHIPS_NAMESPACE = (
- "http://schemas.openxmlformats.org/package/2006/relationships"
- )
- OFFICE_RELATIONSHIPS_NAMESPACE = (
- "http://schemas.openxmlformats.org/officeDocument/2006/relationships"
- )
- CONTENT_TYPES_NAMESPACE = (
- "http://schemas.openxmlformats.org/package/2006/content-types"
- )
- MAIN_CONTENT_FOLDERS = {"word", "ppt", "xl"}
- OOXML_NAMESPACES = {
- "http://schemas.openxmlformats.org/officeDocument/2006/math",
- "http://schemas.openxmlformats.org/officeDocument/2006/relationships",
- "http://schemas.openxmlformats.org/schemaLibrary/2006/main",
- "http://schemas.openxmlformats.org/drawingml/2006/main",
- "http://schemas.openxmlformats.org/drawingml/2006/chart",
- "http://schemas.openxmlformats.org/drawingml/2006/chartDrawing",
- "http://schemas.openxmlformats.org/drawingml/2006/diagram",
- "http://schemas.openxmlformats.org/drawingml/2006/picture",
- "http://schemas.openxmlformats.org/drawingml/2006/spreadsheetDrawing",
- "http://schemas.openxmlformats.org/drawingml/2006/wordprocessingDrawing",
- "http://schemas.openxmlformats.org/wordprocessingml/2006/main",
- "http://schemas.openxmlformats.org/presentationml/2006/main",
- "http://schemas.openxmlformats.org/spreadsheetml/2006/main",
- "http://schemas.openxmlformats.org/officeDocument/2006/sharedTypes",
- "http://www.w3.org/XML/1998/namespace",
- }
- def __init__(self, unpacked_dir, original_file=None, verbose=False):
- self.unpacked_dir = Path(unpacked_dir).resolve()
- self.original_file = Path(original_file) if original_file else None
- self.verbose = verbose
- self.schemas_dir = Path(__file__).parent.parent / "schemas"
- patterns = ["*.xml", "*.rels"]
- self.xml_files = [
- f for pattern in patterns for f in self.unpacked_dir.rglob(pattern)
- ]
- if not self.xml_files:
- print(f"Warning: No XML files found in {self.unpacked_dir}")
- def validate(self):
- raise NotImplementedError("Subclasses must implement the validate method")
- def repair(self) -> int:
- return self.repair_whitespace_preservation()
- def repair_whitespace_preservation(self) -> int:
- repairs = 0
- for xml_file in self.xml_files:
- try:
- content = xml_file.read_text(encoding="utf-8")
- dom = defusedxml.minidom.parseString(content)
- modified = False
- for elem in dom.getElementsByTagName("*"):
- if elem.tagName.endswith(":t") and elem.firstChild:
- text = elem.firstChild.nodeValue
- if text and (text.startswith((' ', '\t')) or text.endswith((' ', '\t'))):
- if elem.getAttribute("xml:space") != "preserve":
- elem.setAttribute("xml:space", "preserve")
- text_preview = repr(text[:30]) + "..." if len(text) > 30 else repr(text)
- print(f" Repaired: {xml_file.name}: Added xml:space='preserve' to {elem.tagName}: {text_preview}")
- repairs += 1
- modified = True
- if modified:
- xml_file.write_bytes(dom.toxml(encoding="UTF-8"))
- except Exception:
- pass
- return repairs
- def validate_xml(self):
- errors = []
- for xml_file in self.xml_files:
- try:
- lxml.etree.parse(str(xml_file))
- except lxml.etree.XMLSyntaxError as e:
- errors.append(
- f" {xml_file.relative_to(self.unpacked_dir)}: "
- f"Line {e.lineno}: {e.msg}"
- )
- except Exception as e:
- errors.append(
- f" {xml_file.relative_to(self.unpacked_dir)}: "
- f"Unexpected error: {str(e)}"
- )
- if errors:
- print(f"FAILED - Found {len(errors)} XML violations:")
- for error in errors:
- print(error)
- return False
- else:
- if self.verbose:
- print("PASSED - All XML files are well-formed")
- return True
- def validate_namespaces(self):
- errors = []
- for xml_file in self.xml_files:
- try:
- root = lxml.etree.parse(str(xml_file)).getroot()
- declared = set(root.nsmap.keys()) - {None}
- for attr_val in [
- v for k, v in root.attrib.items() if k.endswith("Ignorable")
- ]:
- undeclared = set(attr_val.split()) - declared
- errors.extend(
- f" {xml_file.relative_to(self.unpacked_dir)}: "
- f"Namespace '{ns}' in Ignorable but not declared"
- for ns in undeclared
- )
- except lxml.etree.XMLSyntaxError:
- continue
- if errors:
- print(f"FAILED - {len(errors)} namespace issues:")
- for error in errors:
- print(error)
- return False
- if self.verbose:
- print("PASSED - All namespace prefixes properly declared")
- return True
- def validate_unique_ids(self):
- errors = []
- global_ids = {}
- for xml_file in self.xml_files:
- try:
- root = lxml.etree.parse(str(xml_file)).getroot()
- file_ids = {}
- mc_elements = root.xpath(
- ".//mc:AlternateContent", namespaces={"mc": self.MC_NAMESPACE}
- )
- for elem in mc_elements:
- elem.getparent().remove(elem)
- for elem in root.iter():
- tag = (
- elem.tag.split("}")[-1].lower()
- if "}" in elem.tag
- else elem.tag.lower()
- )
- if tag in self.UNIQUE_ID_REQUIREMENTS:
- in_excluded_container = any(
- ancestor.tag.split("}")[-1].lower() in self.EXCLUDED_ID_CONTAINERS
- for ancestor in elem.iterancestors()
- )
- if in_excluded_container:
- continue
- attr_name, scope = self.UNIQUE_ID_REQUIREMENTS[tag]
- id_value = None
- for attr, value in elem.attrib.items():
- attr_local = (
- attr.split("}")[-1].lower()
- if "}" in attr
- else attr.lower()
- )
- if attr_local == attr_name:
- id_value = value
- break
- if id_value is not None:
- if scope == "global":
- if id_value in global_ids:
- prev_file, prev_line, prev_tag = global_ids[
- id_value
- ]
- errors.append(
- f" {xml_file.relative_to(self.unpacked_dir)}: "
- f"Line {elem.sourceline}: Global ID '{id_value}' in <{tag}> "
- f"already used in {prev_file} at line {prev_line} in <{prev_tag}>"
- )
- else:
- global_ids[id_value] = (
- xml_file.relative_to(self.unpacked_dir),
- elem.sourceline,
- tag,
- )
- elif scope == "file":
- key = (tag, attr_name)
- if key not in file_ids:
- file_ids[key] = {}
- if id_value in file_ids[key]:
- prev_line = file_ids[key][id_value]
- errors.append(
- f" {xml_file.relative_to(self.unpacked_dir)}: "
- f"Line {elem.sourceline}: Duplicate {attr_name}='{id_value}' in <{tag}> "
- f"(first occurrence at line {prev_line})"
- )
- else:
- file_ids[key][id_value] = elem.sourceline
- except (lxml.etree.XMLSyntaxError, Exception) as e:
- errors.append(
- f" {xml_file.relative_to(self.unpacked_dir)}: Error: {e}"
- )
- if errors:
- print(f"FAILED - Found {len(errors)} ID uniqueness violations:")
- for error in errors:
- print(error)
- return False
- else:
- if self.verbose:
- print("PASSED - All required IDs are unique")
- return True
- def validate_file_references(self):
- errors = []
- rels_files = list(self.unpacked_dir.rglob("*.rels"))
- if not rels_files:
- if self.verbose:
- print("PASSED - No .rels files found")
- return True
- all_files = []
- for file_path in self.unpacked_dir.rglob("*"):
- if (
- file_path.is_file()
- and file_path.name != "[Content_Types].xml"
- and not file_path.name.endswith(".rels")
- ):
- all_files.append(file_path.resolve())
- all_referenced_files = set()
- if self.verbose:
- print(
- f"Found {len(rels_files)} .rels files and {len(all_files)} target files"
- )
- for rels_file in rels_files:
- try:
- rels_root = lxml.etree.parse(str(rels_file)).getroot()
- rels_dir = rels_file.parent
- referenced_files = set()
- broken_refs = []
- for rel in rels_root.findall(
- ".//ns:Relationship",
- namespaces={"ns": self.PACKAGE_RELATIONSHIPS_NAMESPACE},
- ):
- target = rel.get("Target")
- if target and not target.startswith(
- ("http", "mailto:")
- ):
- if target.startswith("/"):
- target_path = self.unpacked_dir / target.lstrip("/")
- elif rels_file.name == ".rels":
- target_path = self.unpacked_dir / target
- else:
- base_dir = rels_dir.parent
- target_path = base_dir / target
- try:
- target_path = target_path.resolve()
- if target_path.exists() and target_path.is_file():
- referenced_files.add(target_path)
- all_referenced_files.add(target_path)
- else:
- broken_refs.append((target, rel.sourceline))
- except (OSError, ValueError):
- broken_refs.append((target, rel.sourceline))
- if broken_refs:
- rel_path = rels_file.relative_to(self.unpacked_dir)
- for broken_ref, line_num in broken_refs:
- errors.append(
- f" {rel_path}: Line {line_num}: Broken reference to {broken_ref}"
- )
- except Exception as e:
- rel_path = rels_file.relative_to(self.unpacked_dir)
- errors.append(f" Error parsing {rel_path}: {e}")
- unreferenced_files = set(all_files) - all_referenced_files
- if unreferenced_files:
- for unref_file in sorted(unreferenced_files):
- unref_rel_path = unref_file.relative_to(self.unpacked_dir)
- errors.append(f" Unreferenced file: {unref_rel_path}")
- if errors:
- print(f"FAILED - Found {len(errors)} relationship validation errors:")
- for error in errors:
- print(error)
- print(
- "CRITICAL: These errors will cause the document to appear corrupt. "
- + "Broken references MUST be fixed, "
- + "and unreferenced files MUST be referenced or removed."
- )
- return False
- else:
- if self.verbose:
- print(
- "PASSED - All references are valid and all files are properly referenced"
- )
- return True
- def validate_all_relationship_ids(self):
- import lxml.etree
- errors = []
- for xml_file in self.xml_files:
- if xml_file.suffix == ".rels":
- continue
- rels_dir = xml_file.parent / "_rels"
- rels_file = rels_dir / f"{xml_file.name}.rels"
- if not rels_file.exists():
- continue
- try:
- rels_root = lxml.etree.parse(str(rels_file)).getroot()
- rid_to_type = {}
- for rel in rels_root.findall(
- f".//{{{self.PACKAGE_RELATIONSHIPS_NAMESPACE}}}Relationship"
- ):
- rid = rel.get("Id")
- rel_type = rel.get("Type", "")
- if rid:
- if rid in rid_to_type:
- rels_rel_path = rels_file.relative_to(self.unpacked_dir)
- errors.append(
- f" {rels_rel_path}: Line {rel.sourceline}: "
- f"Duplicate relationship ID '{rid}' (IDs must be unique)"
- )
- type_name = (
- rel_type.split("/")[-1] if "/" in rel_type else rel_type
- )
- rid_to_type[rid] = type_name
- xml_root = lxml.etree.parse(str(xml_file)).getroot()
- r_ns = self.OFFICE_RELATIONSHIPS_NAMESPACE
- rid_attrs_to_check = ["id", "embed", "link"]
- for elem in xml_root.iter():
- for attr_name in rid_attrs_to_check:
- rid_attr = elem.get(f"{{{r_ns}}}{attr_name}")
- if not rid_attr:
- continue
- xml_rel_path = xml_file.relative_to(self.unpacked_dir)
- elem_name = (
- elem.tag.split("}")[-1] if "}" in elem.tag else elem.tag
- )
- if rid_attr not in rid_to_type:
- errors.append(
- f" {xml_rel_path}: Line {elem.sourceline}: "
- f"<{elem_name}> r:{attr_name} references non-existent relationship '{rid_attr}' "
- f"(valid IDs: {', '.join(sorted(rid_to_type.keys())[:5])}{'...' if len(rid_to_type) > 5 else ''})"
- )
- elif attr_name == "id" and self.ELEMENT_RELATIONSHIP_TYPES:
- expected_type = self._get_expected_relationship_type(
- elem_name
- )
- if expected_type:
- actual_type = rid_to_type[rid_attr]
- if expected_type not in actual_type.lower():
- errors.append(
- f" {xml_rel_path}: Line {elem.sourceline}: "
- f"<{elem_name}> references '{rid_attr}' which points to '{actual_type}' "
- f"but should point to a '{expected_type}' relationship"
- )
- except Exception as e:
- xml_rel_path = xml_file.relative_to(self.unpacked_dir)
- errors.append(f" Error processing {xml_rel_path}: {e}")
- if errors:
- print(f"FAILED - Found {len(errors)} relationship ID reference errors:")
- for error in errors:
- print(error)
- print("\nThese ID mismatches will cause the document to appear corrupt!")
- return False
- else:
- if self.verbose:
- print("PASSED - All relationship ID references are valid")
- return True
- def _get_expected_relationship_type(self, element_name):
- elem_lower = element_name.lower()
- if elem_lower in self.ELEMENT_RELATIONSHIP_TYPES:
- return self.ELEMENT_RELATIONSHIP_TYPES[elem_lower]
- if elem_lower.endswith("id") and len(elem_lower) > 2:
- prefix = elem_lower[:-2]
- if prefix.endswith("master"):
- return prefix.lower()
- elif prefix.endswith("layout"):
- return prefix.lower()
- else:
- if prefix == "sld":
- return "slide"
- return prefix.lower()
- if elem_lower.endswith("reference") and len(elem_lower) > 9:
- prefix = elem_lower[:-9]
- return prefix.lower()
- return None
- def validate_content_types(self):
- errors = []
- content_types_file = self.unpacked_dir / "[Content_Types].xml"
- if not content_types_file.exists():
- print("FAILED - [Content_Types].xml file not found")
- return False
- try:
- root = lxml.etree.parse(str(content_types_file)).getroot()
- declared_parts = set()
- declared_extensions = set()
- for override in root.findall(
- f".//{{{self.CONTENT_TYPES_NAMESPACE}}}Override"
- ):
- part_name = override.get("PartName")
- if part_name is not None:
- declared_parts.add(part_name.lstrip("/"))
- for default in root.findall(
- f".//{{{self.CONTENT_TYPES_NAMESPACE}}}Default"
- ):
- extension = default.get("Extension")
- if extension is not None:
- declared_extensions.add(extension.lower())
- declarable_roots = {
- "sld",
- "sldLayout",
- "sldMaster",
- "presentation",
- "document",
- "workbook",
- "worksheet",
- "theme",
- }
- media_extensions = {
- "png": "image/png",
- "jpg": "image/jpeg",
- "jpeg": "image/jpeg",
- "gif": "image/gif",
- "bmp": "image/bmp",
- "tiff": "image/tiff",
- "wmf": "image/x-wmf",
- "emf": "image/x-emf",
- }
- all_files = list(self.unpacked_dir.rglob("*"))
- all_files = [f for f in all_files if f.is_file()]
- for xml_file in self.xml_files:
- path_str = str(xml_file.relative_to(self.unpacked_dir)).replace(
- "\\", "/"
- )
- if any(
- skip in path_str
- for skip in [".rels", "[Content_Types]", "docProps/", "_rels/"]
- ):
- continue
- try:
- root_tag = lxml.etree.parse(str(xml_file)).getroot().tag
- root_name = root_tag.split("}")[-1] if "}" in root_tag else root_tag
- if root_name in declarable_roots and path_str not in declared_parts:
- errors.append(
- f" {path_str}: File with <{root_name}> root not declared in [Content_Types].xml"
- )
- except Exception:
- continue
- for file_path in all_files:
- if file_path.suffix.lower() in {".xml", ".rels"}:
- continue
- if file_path.name == "[Content_Types].xml":
- continue
- if "_rels" in file_path.parts or "docProps" in file_path.parts:
- continue
- extension = file_path.suffix.lstrip(".").lower()
- if extension and extension not in declared_extensions:
- if extension in media_extensions:
- relative_path = file_path.relative_to(self.unpacked_dir)
- errors.append(
- f' {relative_path}: File with extension \'{extension}\' not declared in [Content_Types].xml - should add: <Default Extension="{extension}" ContentType="{media_extensions[extension]}"/>'
- )
- except Exception as e:
- errors.append(f" Error parsing [Content_Types].xml: {e}")
- if errors:
- print(f"FAILED - Found {len(errors)} content type declaration errors:")
- for error in errors:
- print(error)
- return False
- else:
- if self.verbose:
- print(
- "PASSED - All content files are properly declared in [Content_Types].xml"
- )
- return True
- def validate_file_against_xsd(self, xml_file, verbose=False):
- xml_file = Path(xml_file).resolve()
- unpacked_dir = self.unpacked_dir.resolve()
- is_valid, current_errors = self._validate_single_file_xsd(
- xml_file, unpacked_dir
- )
- if is_valid is None:
- return None, set()
- elif is_valid:
- return True, set()
- original_errors = self._get_original_file_errors(xml_file)
- assert current_errors is not None
- new_errors = current_errors - original_errors
- new_errors = {
- e for e in new_errors
- if not any(pattern in e for pattern in self.IGNORED_VALIDATION_ERRORS)
- }
- if new_errors:
- if verbose:
- relative_path = xml_file.relative_to(unpacked_dir)
- print(f"FAILED - {relative_path}: {len(new_errors)} new error(s)")
- for error in list(new_errors)[:3]:
- truncated = error[:250] + "..." if len(error) > 250 else error
- print(f" - {truncated}")
- return False, new_errors
- else:
- if verbose:
- print(
- f"PASSED - No new errors (original had {len(current_errors)} errors)"
- )
- return True, set()
- def validate_against_xsd(self):
- new_errors = []
- original_error_count = 0
- valid_count = 0
- skipped_count = 0
- for xml_file in self.xml_files:
- relative_path = str(xml_file.relative_to(self.unpacked_dir))
- is_valid, new_file_errors = self.validate_file_against_xsd(
- xml_file, verbose=False
- )
- if is_valid is None:
- skipped_count += 1
- continue
- elif is_valid and not new_file_errors:
- valid_count += 1
- continue
- elif is_valid:
- original_error_count += 1
- valid_count += 1
- continue
- new_errors.append(f" {relative_path}: {len(new_file_errors)} new error(s)")
- for error in list(new_file_errors)[:3]:
- new_errors.append(
- f" - {error[:250]}..." if len(error) > 250 else f" - {error}"
- )
- if self.verbose:
- print(f"Validated {len(self.xml_files)} files:")
- print(f" - Valid: {valid_count}")
- print(f" - Skipped (no schema): {skipped_count}")
- if original_error_count:
- print(f" - With original errors (ignored): {original_error_count}")
- print(
- f" - With NEW errors: {len(new_errors) > 0 and len([e for e in new_errors if not e.startswith(' ')]) or 0}"
- )
- if new_errors:
- print("\nFAILED - Found NEW validation errors:")
- for error in new_errors:
- print(error)
- return False
- else:
- if self.verbose:
- print("\nPASSED - No new XSD validation errors introduced")
- return True
- def _get_schema_path(self, xml_file):
- if xml_file.name in self.SCHEMA_MAPPINGS:
- return self.schemas_dir / self.SCHEMA_MAPPINGS[xml_file.name]
- if xml_file.suffix == ".rels":
- return self.schemas_dir / self.SCHEMA_MAPPINGS[".rels"]
- if "charts/" in str(xml_file) and xml_file.name.startswith("chart"):
- return self.schemas_dir / self.SCHEMA_MAPPINGS["chart"]
- if "theme/" in str(xml_file) and xml_file.name.startswith("theme"):
- return self.schemas_dir / self.SCHEMA_MAPPINGS["theme"]
- if xml_file.parent.name in self.MAIN_CONTENT_FOLDERS:
- return self.schemas_dir / self.SCHEMA_MAPPINGS[xml_file.parent.name]
- return None
- def _clean_ignorable_namespaces(self, xml_doc):
- xml_string = lxml.etree.tostring(xml_doc, encoding="unicode")
- xml_copy = lxml.etree.fromstring(xml_string)
- for elem in xml_copy.iter():
- attrs_to_remove = []
- for attr in elem.attrib:
- if "{" in attr:
- ns = attr.split("}")[0][1:]
- if ns not in self.OOXML_NAMESPACES:
- attrs_to_remove.append(attr)
- for attr in attrs_to_remove:
- del elem.attrib[attr]
- self._remove_ignorable_elements(xml_copy)
- return lxml.etree.ElementTree(xml_copy)
- def _remove_ignorable_elements(self, root):
- elements_to_remove = []
- for elem in list(root):
- if not hasattr(elem, "tag") or callable(elem.tag):
- continue
- tag_str = str(elem.tag)
- if tag_str.startswith("{"):
- ns = tag_str.split("}")[0][1:]
- if ns not in self.OOXML_NAMESPACES:
- elements_to_remove.append(elem)
- continue
- self._remove_ignorable_elements(elem)
- for elem in elements_to_remove:
- root.remove(elem)
- def _preprocess_for_mc_ignorable(self, xml_doc):
- root = xml_doc.getroot()
- if f"{{{self.MC_NAMESPACE}}}Ignorable" in root.attrib:
- del root.attrib[f"{{{self.MC_NAMESPACE}}}Ignorable"]
- return xml_doc
- def _validate_single_file_xsd(self, xml_file, base_path):
- schema_path = self._get_schema_path(xml_file)
- if not schema_path:
- return None, None
- try:
- with open(schema_path, "rb") as xsd_file:
- parser = lxml.etree.XMLParser()
- xsd_doc = lxml.etree.parse(
- xsd_file, parser=parser, base_url=str(schema_path)
- )
- schema = lxml.etree.XMLSchema(xsd_doc)
- with open(xml_file, "r") as f:
- xml_doc = lxml.etree.parse(f)
- xml_doc, _ = self._remove_template_tags_from_text_nodes(xml_doc)
- xml_doc = self._preprocess_for_mc_ignorable(xml_doc)
- relative_path = xml_file.relative_to(base_path)
- if (
- relative_path.parts
- and relative_path.parts[0] in self.MAIN_CONTENT_FOLDERS
- ):
- xml_doc = self._clean_ignorable_namespaces(xml_doc)
- if schema.validate(xml_doc):
- return True, set()
- else:
- errors = set()
- for error in schema.error_log:
- errors.add(error.message)
- return False, errors
- except Exception as e:
- return False, {str(e)}
- def _get_original_file_errors(self, xml_file):
- if self.original_file is None:
- return set()
- import tempfile
- import zipfile
- xml_file = Path(xml_file).resolve()
- unpacked_dir = self.unpacked_dir.resolve()
- relative_path = xml_file.relative_to(unpacked_dir)
- with tempfile.TemporaryDirectory() as temp_dir:
- temp_path = Path(temp_dir)
- with zipfile.ZipFile(self.original_file, "r") as zip_ref:
- zip_ref.extractall(temp_path)
- original_xml_file = temp_path / relative_path
- if not original_xml_file.exists():
- return set()
- is_valid, errors = self._validate_single_file_xsd(
- original_xml_file, temp_path
- )
- return errors if errors else set()
- def _remove_template_tags_from_text_nodes(self, xml_doc):
- warnings = []
- template_pattern = re.compile(r"\{\{[^}]*\}\}")
- xml_string = lxml.etree.tostring(xml_doc, encoding="unicode")
- xml_copy = lxml.etree.fromstring(xml_string)
- def process_text_content(text, content_type):
- if not text:
- return text
- matches = list(template_pattern.finditer(text))
- if matches:
- for match in matches:
- warnings.append(
- f"Found template tag in {content_type}: {match.group()}"
- )
- return template_pattern.sub("", text)
- return text
- for elem in xml_copy.iter():
- if not hasattr(elem, "tag") or callable(elem.tag):
- continue
- tag_str = str(elem.tag)
- if tag_str.endswith("}t") or tag_str == "t":
- continue
- elem.text = process_text_content(elem.text, "text content")
- elem.tail = process_text_content(elem.tail, "tail content")
- return lxml.etree.ElementTree(xml_copy), warnings
- if __name__ == "__main__":
- raise RuntimeError("This module should not be run directly.")
|