почистил ссылки в сипиське источников (там почему-то стояли гиперссылки)

This commit is contained in:
2026-05-31 20:29:59 +03:00
parent b84506d8d2
commit 8508150b07
2 changed files with 192 additions and 0 deletions
+192
View File
@@ -0,0 +1,192 @@
#!/usr/bin/env python3
import re
import shutil
import tempfile
import xml.etree.ElementTree as ET
from pathlib import Path
from zipfile import ZipFile
FIRST_NEW_SOURCE = 8
NS = {
"w": "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
}
def normalize_url(url):
url = url.strip()
if "#:~:text=" in url:
url = url.split("#:~:text=", 1)[0]
url = url.rstrip("/")
return url
def build_url_mapping(root):
"""
Первый проход.
Собираем все уникальные URL и
присваиваем им номера.
"""
url_to_index = {}
next_index = FIRST_NEW_SOURCE
for node in root.findall(".//w:instrText", NS):
text = "".join(node.itertext())
match = re.search(
r'HYPERLINK\s+"([^"]+)"',
text
)
if not match:
continue
url = normalize_url(match.group(1))
if url in url_to_index:
continue
url_to_index[url] = next_index
next_index += 1
return url_to_index
def patch_document(root, url_to_index):
"""
Второй проход.
Находим:
HYPERLINK "..."
Затем ближайший следующий
<w:t>[старый номер]</w:t>
И меняем его на новый.
"""
current_url = None
waiting_for_visible_text = False
for elem in root.iter():
tag = elem.tag
if tag.endswith("instrText"):
text = "".join(elem.itertext())
match = re.search(
r'HYPERLINK\s+"([^"]+)"',
text
)
if match:
current_url = normalize_url(
match.group(1)
)
waiting_for_visible_text = True
continue
if (
waiting_for_visible_text
and tag.endswith("t")
and elem.text
and re.match(r"\[\d+\]", elem.text.strip())
):
index = url_to_index[current_url]
old = elem.text
elem.text = f"[{index}]"
print(
f"{old} -> {elem.text} : {current_url}"
)
waiting_for_visible_text = False
current_url = None
def main():
input_docx = Path("Отчет.docx")
output_docx = Path(
input_docx.stem + "_fixed.docx"
)
with tempfile.TemporaryDirectory() as tmp:
tmp = Path(tmp)
with ZipFile(input_docx) as archive:
archive.extractall(tmp)
document_xml = (
tmp /
"word" /
"document.xml"
)
tree = ET.parse(document_xml)
root = tree.getroot()
url_to_index = build_url_mapping(root)
print("\nURL -> INDEX\n")
for url, index in url_to_index.items():
print(
f"[{index}] {url}"
)
print(
f"\nFound {len(url_to_index)} unique sources\n"
)
patch_document(
root,
url_to_index
)
tree.write(
document_xml,
encoding="utf-8",
xml_declaration=True
)
with ZipFile(
output_docx,
"w"
) as out_zip:
for file in tmp.rglob("*"):
if file.is_dir():
continue
out_zip.write(
file,
file.relative_to(tmp)
)
print(
f"\nSaved: {output_docx}"
)
if __name__ == "__main__":
main()
BIN
View File
Binary file not shown.