Files

46 lines
1.5 KiB
Python

"""
JIRA's Description field comes back as Atlassian Document Format (ADF) -
a nested JSON tree, not plain text. This walks it and extracts readable
text, since all we need here is "what did the staff member write about
what boxes are needed", not full document fidelity.
Degrades gracefully on node types it doesn't know (tables, mentions,
emoji, etc.) rather than raising - a partially-extracted description is
far more useful than a crash on an edge case we didn't anticipate.
"""
from __future__ import annotations
from typing import Optional
# Block-level node types that should end with a line break once their
# content has been extracted, so paragraphs/list items don't run together.
_BLOCK_TYPES = {"paragraph", "listItem", "heading", "codeBlock", "blockquote"}
def adf_to_text(adf: Optional[dict]) -> str:
if not isinstance(adf, dict):
return ""
lines: list[str] = []
_walk(adf, lines)
# Collapse the accumulated fragments, trim stray blank lines from
# nested block boundaries.
text = "".join(lines)
return "\n".join(line.rstrip() for line in text.split("\n")).strip()
def _walk(node: dict, lines: list[str]) -> None:
node_type = node.get("type")
if node_type == "text":
lines.append(node.get("text", ""))
elif node_type == "hardBreak":
lines.append("\n")
for child in node.get("content", []) or []:
if isinstance(child, dict):
_walk(child, lines)
if node_type in _BLOCK_TYPES:
lines.append("\n")