У меня есть этот код, который принимает файл XML, берет дочерние элементы (тег text
) тега new_line
и сохраняет их индекс как ключ в словаре, а элементы как значения в том же самом толковый словарь. Я хочу удалить ключи словаря, значения которых содержат «10.238», но, похоже, это не работает. Все остальное работает. Это мой код:
import re
from xml.dom import minidom
from xml.etree import ElementTree as ET
def filter_values_by_keyword(my_dict, filter_by):
"""
Return a list of values which contains `filter_by` keyword.
Arguments:
my_dict (dict): Dict containing (...data specifics here)
filter_by (str): Keyword to look for in values of my_dict
Return:
List of filtered values
"""
return [key for key, value in my_dict.items() if filter_by in value]
def get_xml_by_tag_names(xml_path, tag_name_1, tag_name_2):
"""
Your docstring here.
"""
data = {}
xml_tree = minidom.parse(xml_path)
item_group_nodes = xml_tree.getElementsByTagName(tag_name_1)
for idx, item_group_node in enumerate(item_group_nodes):
cl_compile_nodes = item_group_node.getElementsByTagName(tag_name_2)
for _ in cl_compile_nodes:
data[idx]=[item_group_node.toxml()]
return data
def main():
data = get_xml_by_tag_names('output2.xml', 'new_line', 'text')
filtered_values = filter_values_by_keyword(data, '10.238')
for item in filtered_values:
del data[item]
for value in data.values():
myxml = ' '.join(value)
# print(myxml)
tree = ET.fromstring(myxml)
lista = ([text.text for text in tree.findall('text')])
testo = (' '.join(lista))
print(testo)
if __name__ == "__main__":
main()
А это образец XML:
<pages>
<page id="1" bbox="0.000,0.000,462.047,680.315" rotate="0">
<textbox id="0" bbox="191.745,592.218,249.042,603.578">
<textline>
<new_line>
<text font="NUMPTY+ImprintMTnum" bbox="297.284,540.828,300.188,553.310" colourspace="DeviceGray" ncolour="0" size="12.482">della quale non conosce che una parte;] </text>
<text font="PYNIYO+ImprintMTnum-Italic" bbox="322.455,540.839,328.251,553.566" colourspace="DeviceGray" ncolour="0" size="12.727">prima</text>
<text font="NUMPTY+ImprintMTnum" bbox="331.206,545.345,334.683,552.834" colourspace="DeviceGray" ncolour="0" size="7.489">1</text>
<text font="NUMPTY+ImprintMTnum" bbox="177.602,528.028,180.850,540.510" colourspace="DeviceGray" ncolour="0" size="12.482">che nonconosce ancora appieno;</text>
<text font="NUMPTY+ImprintMTnum" bbox="189.430,532.545,192.908,540.034" colourspace="DeviceGray" ncolour="0" size="7.489">2</text>
<text font="NUMPTY+ImprintMTnum" bbox="203.879,528.028,208.975,540.510" colourspace="DeviceGray" ncolour="0" size="12.482">che</text>
</new_line>
</textline>
<textline bbox="68.032,408.428,372.762,421.166">
<new_line>
<text font="NUMPTY+ImprintMTnum" bbox="307.143,408.428,310.392,420.910" colourspace="DeviceGray" ncolour="0" size="12.482">viso] vi</text>
<text font="NUMPTY+ImprintMTnum" bbox="310.280,408.808,313.243,419.046" colourspace="DeviceGray" ncolour="0" size="10.238">-</text>
<text font="PYNIYO+ImprintMTnum-Italic" bbox="320.072,408.439,325.868,421.166" colourspace="DeviceGray" ncolour="0" size="12.727">su</text>
<text font="NUMPTY+ImprintMTnum" bbox="328.829,408.428,338.452,420.910" colourspace="DeviceGray" ncolour="0" size="12.482">m</text>
</new_line>
</textline>
</textbox>
</page>
</pages>