INNER CODE UNIT · Python
_asset_desc
Laurent00TT/PharosRAG · src/chunker/core.py:69
def _asset_desc(s, cap=800):
"""An image's VLM-extracted content (OCR text / mermaid / chart data) or alt-text, cleaned for
use as retrieval text. For mermaid (flowchart/diagram), keep only the node + edge LABELS — the
real retrieval signal — and drop scaffolding (`graph TD`, `-->`, `style X fill:#f9f`), which is
zero-value noise in the embedding (review F2). Cap at a word boundary, not mid-token."""
if not s:
return ""
s = re.sub(r"```[a-z]*|```", " ", s)
if re.search(r"\bgraph\s+(?:TD|TB|LR|RL|BT)\b|\bflowchart\b|--?>", s):
labels = re.findall(r'[\[\(\{|]\s*"?([^"\[\]\(\)\{\}|]+?)"?\s*[\]\)\}|]', s)
s = " ".join(x.strip() for x in labels if x.strip())
s = re.sub(r"\s+", " ", s).strip()
return s[:cap].rsplit(" ", 1)[0] if len(s) > cap else s
_TR_RE = re.compile(r"<tr[^>]*>(.*?)</tr>", re.S | re.I)
_CELL_RE = re.compile(r"<t[dh][^>]*>(.*?)</t[dh]>", re.S | re.I)
_TAG_RE = re.compile(r"<[^>]+>")