Find tables inside HTML and extract the contents into objects using the following logic:
- If table has a single column, just create an array of strings from the values
- If table has 2 columns and has no header row, treat the first column as key and second as value and create a table of key/value
- If table has a header row, create a table of objects where attribute names are the headers
- If table does not have a header row, create table of objects where attribute names are cell1, cell2, cell3...
import demistomock as demisto # noqa: F401
from bs4 import BeautifulSoup
from CommonServerPython import * # noqa: F401
def extract_html_table(html, indexes):
soup = BeautifulSoup(html, "html.parser")
tables = []
for index, tab in enumerate(soup.find_all("table")):
if len(indexes) > 0 and index not in indexes and str(index) not in indexes:
continue
table: Union[list, dict] = []
headers = []
# Check if there are headers and use them
try:
for th in tab.find_all("th"): # type: ignore
headers.append(th.text)
for tr in tab.find_all("tr"): # type: ignore
tds = tr.find_all("td")
# This is a data row and not header row
if len(tds) > 0:
# Single value in a table - just create an array of strings ignoring header
if len(tds) == 1:
table.append(tds[0].text) # type: ignore[union-attr]
# If there are 2 columns and no headers,
# treat as key-value (might override values if same key in first column)
elif len(tds) == 2 and len(headers) == 0:
if type(table) is list:
table = {} # type: ignore
table[tds[0].text] = tds[1].text # type: ignore[call-overload]
else:
row = {}
if len(headers) > 0:
for i, td in enumerate(tds):
row[headers[i]] = td.text
else:
for i, td in enumerate(tds):
row["cell" + str(i)] = td.text
if isinstance(table, list): # type: ignore[arg-type]
table.append(row)
except Exception as e:
demisto.debug(f"Failed to extract table: {e}")
if len(table) > 0:
tables.append(table)
if len(tables) > 0:
return {
"Type": entryTypes["note"],
"Contents": f"Found {len(tables)} tables in HTML.",
"ContentsFormat": formats["text"],
"EntryContext": {"HTMLTables": tables if len(tables) > 1 else tables[0]},
}
else:
return "Did not find tables in HTML."
def main():
html = demisto.getArg("html")
indexes = argToList(demisto.getArg("indexes"))
demisto.results(extract_html_table(html, indexes))
if __name__ in ["__main__", "builtin", "builtins"]:
main()
README
Find tables inside HTML and extract the contents into objects using the following logic:
If the table has a single column, just create an array of strings from the values.
If the table has 2 columns and has no header row, treat the first column as the key and the second column as the value and create a table for the key/value.
If the table has a header row, create a table of objects where the attribute names are the headers.
If the table does not have a header row, create table of objects where attribute names are cell1, cell2, cell3…
Script Data
Name
Description
Script Type
python
Tags
Utility
Inputs
Argument Name
Description
html
The HTML to extract the tables from.
indexes
Extracts only the tables with given indexes. IT will be, 0 based.