I want to extract data from a table while preserving everything including text, MathType equations, images, and all other content. Could you help me with that?
I already have the code as shown below, but it does not work as expected.
def copy_table_rows_with_cloned_paragraphs(file_path, output_file):
# Load the source document
doc = Document()
doc.LoadFromFile(file_path)
# Create a new document for output
other_doc = Document()
# Iterate through sections in the document
for section_index in range(doc.Sections.Count):
section = doc.Sections.get_Item(section_index) # Access each section
new_section = other_doc.AddSection() # Add a new section to the new document
# Iterate through tables in the section
for table_index in range(section.Tables.Count):
table = section.Tables.get_Item(table_index) # Access each table in the section
# Iterate through rows in the table
for row_index in range(table.Rows.Count):
row = table.Rows.get_Item(row_index) # Access each row
# Iterate through the cells in the row
for cell_index in range(row.Cells.Count):
cell = row.Cells.get_Item(cell_index) # Access each cell
for para_index in range(cell.Paragraphs.Count):
paragraph = cell.Paragraphs.get_Item(para_index) # Access each paragraph
new_paragraph = new_section.Body.AddParagraph() # Add new paragraph to the new section
# Clone the paragraph (content + format + all internal elements)
new_paragraph = paragraph.Clone() # Clone the entire paragraph (including all internal content)
# Save the new document
other_doc.SaveToFile(output_file)
print(f"Document with cloned paragraphs copied successfully to: {output_file}")