Spaces:
Running on CPU Upgrade
Running on CPU Upgrade
| from __future__ import annotations | |
| import json | |
| import re | |
| from typing import Any | |
| import bleach | |
| import markdown | |
| TABLE_RE = re.compile(r"<table\b[\s\S]*?</table>", re.IGNORECASE) | |
| JSON_OBJECT_RE = re.compile( | |
| r'\{[^{}]*?"bbox"\s*:\s*\[[^\]]+\][^{}]*?\}', | |
| re.IGNORECASE, | |
| ) | |
| BBOX_TAG_RE = re.compile( | |
| r"""<(?:bbox|box)(?:\s+label=["']?([^"'>]+)["']?)?[^>]*> | |
| \s*\[?\s*([\d.]+)\s*,\s*([\d.]+)\s*,\s*([\d.]+)\s*,\s*([\d.]+)\s*\]? | |
| \s*</(?:bbox|box)>""", | |
| re.IGNORECASE | re.VERBOSE, | |
| ) | |
| TABLE_TAGS = [ | |
| "table", | |
| "caption", | |
| "thead", | |
| "tbody", | |
| "tfoot", | |
| "tr", | |
| "th", | |
| "td", | |
| "br", | |
| "strong", | |
| "em", | |
| ] | |
| TABLE_ATTRIBUTES = { | |
| "th": ["colspan", "rowspan", "scope"], | |
| "td": ["colspan", "rowspan"], | |
| } | |
| TEXT_TAGS = [ | |
| "p", | |
| "h1", | |
| "h2", | |
| "h3", | |
| "h4", | |
| "h5", | |
| "ul", | |
| "ol", | |
| "li", | |
| "strong", | |
| "em", | |
| "code", | |
| "pre", | |
| "blockquote", | |
| "hr", | |
| "br", | |
| ] | |
| TABLE_TEXT_MARKER = "`<table>...</table>`" | |
| def _normalize_box(candidate: dict[str, Any], index: int) -> dict | None: | |
| bbox = candidate.get("bbox") | |
| if not isinstance(bbox, list) or len(bbox) != 4: | |
| return None | |
| try: | |
| coords = [float(value) for value in bbox] | |
| except (TypeError, ValueError): | |
| return None | |
| maximum = max(coords) | |
| scale = 1 if maximum <= 1 else (1000 if maximum <= 1000 else maximum) | |
| x1, y1, x2, y2 = [min(1, max(0, value / scale)) for value in coords] | |
| if x2 <= x1 or y2 <= y1: | |
| return None | |
| label = str(candidate.get("label") or candidate.get("type") or "element") | |
| return { | |
| "id": f"{index}-{'-'.join(str(value) for value in coords)}", | |
| "label": label[:48], | |
| "bbox": [x1, y1, x2, y2], | |
| } | |
| def extract_bounding_boxes(raw_text: str) -> list[dict]: | |
| candidates: list[dict] = [] | |
| for match in JSON_OBJECT_RE.finditer(raw_text): | |
| try: | |
| candidate = json.loads(match.group(0)) | |
| except json.JSONDecodeError: | |
| continue | |
| if isinstance(candidate, dict): | |
| candidates.append(candidate) | |
| for match in BBOX_TAG_RE.finditer(raw_text): | |
| candidates.append( | |
| { | |
| "label": match.group(1) or "element", | |
| "bbox": [float(match.group(index)) for index in range(2, 6)], | |
| } | |
| ) | |
| boxes = [] | |
| for index, candidate in enumerate(candidates): | |
| normalized = _normalize_box(candidate, index) | |
| if normalized: | |
| boxes.append(normalized) | |
| return boxes[:250] | |
| def extract_tables(raw_text: str) -> str: | |
| tables = TABLE_RE.findall(raw_text) | |
| if not tables: | |
| return "" | |
| return bleach.clean( | |
| "\n".join(tables), | |
| tags=TABLE_TAGS, | |
| attributes=TABLE_ATTRIBUTES, | |
| protocols=[], | |
| strip=True, | |
| ) | |
| def render_text(raw_text: str) -> str: | |
| text_with_table_markers = TABLE_RE.sub( | |
| f"\n\n{TABLE_TEXT_MARKER}\n\n", | |
| raw_text, | |
| ) | |
| rendered = markdown.markdown( | |
| text_with_table_markers, | |
| extensions=["extra", "sane_lists"], | |
| output_format="html", | |
| ) | |
| return bleach.clean( | |
| rendered, | |
| tags=TEXT_TAGS, | |
| attributes={}, | |
| protocols=[], | |
| strip=True, | |
| ) | |
| def shape_parser_output(raw_text: str) -> dict: | |
| output = raw_text if isinstance(raw_text, str) else "" | |
| return { | |
| "raw_output": output, | |
| "output_chars": len(output), | |
| "boxes": extract_bounding_boxes(output), | |
| "tables_html": extract_tables(output), | |
| "text_html": render_text(output), | |
| } | |
| def _structured_box( | |
| payload: dict[str, Any], | |
| *, | |
| label: str, | |
| index: int, | |
| ) -> dict | None: | |
| bbox = payload.get("bounding_box_normalized") | |
| if not isinstance(bbox, dict): | |
| return None | |
| return _normalize_box( | |
| { | |
| "label": label, | |
| "bbox": [ | |
| bbox.get("top_left_x"), | |
| bbox.get("top_left_y"), | |
| bbox.get("bottom_right_x"), | |
| bbox.get("bottom_right_y"), | |
| ], | |
| }, | |
| index, | |
| ) | |
| def _shape_blocks_page(page: dict[str, Any]) -> tuple[str, list[dict]]: | |
| blocks = page.get("blocks") | |
| if not isinstance(blocks, list): | |
| raise ValueError("blocks page is missing blocks") | |
| content: list[str] = [] | |
| boxes: list[dict] = [] | |
| for block in blocks: | |
| if not isinstance(block, dict): | |
| raise ValueError("invalid parse block") | |
| block_type = block.get("type") | |
| if block_type == "text": | |
| text = block.get("text") | |
| if not isinstance(text, dict) or not isinstance(text.get("content"), str): | |
| raise ValueError("invalid text block") | |
| content.append(text["content"]) | |
| continue | |
| if block_type == "table": | |
| table = block.get("table") | |
| if not isinstance(table, dict) or not isinstance(table.get("html"), str): | |
| raise ValueError("invalid table block") | |
| content.append(table["html"]) | |
| box = _structured_box( | |
| table, | |
| label=str(table.get("title") or "table"), | |
| index=len(boxes), | |
| ) | |
| elif block_type == "image": | |
| image = block.get("image") | |
| if not isinstance(image, dict): | |
| raise ValueError("invalid image block") | |
| description = str(image.get("description") or "Image") | |
| content.append(f"*Image: {description}*") | |
| box = _structured_box( | |
| image, | |
| label=str(image.get("category") or image.get("id") or "image"), | |
| index=len(boxes), | |
| ) | |
| else: | |
| raise ValueError("unsupported parse block") | |
| if box: | |
| boxes.append(box) | |
| return "\n\n".join(content), boxes | |
| def _shape_markdown_page(page: dict[str, Any]) -> tuple[str, list[dict]]: | |
| markdown_page = page.get("markdown") | |
| if not isinstance(markdown_page, dict) or not isinstance( | |
| markdown_page.get("content"), str | |
| ): | |
| raise ValueError("invalid markdown page") | |
| boxes = [] | |
| for image in markdown_page.get("images") or []: | |
| if not isinstance(image, dict): | |
| raise ValueError("invalid markdown image") | |
| box = _structured_box( | |
| image, | |
| label=str(image.get("category") or image.get("id") or "image"), | |
| index=len(boxes), | |
| ) | |
| if box: | |
| boxes.append(box) | |
| return markdown_page["content"], boxes | |
| def shape_parse_response(body: dict[str, Any]) -> dict: | |
| if not isinstance(body, dict): | |
| raise TypeError("parse response must be an object") | |
| pages = body.get("pages") | |
| if not isinstance(pages, list) or not pages: | |
| raise ValueError("parse response is missing pages") | |
| page_content: list[str] = [] | |
| boxes: list[dict] = [] | |
| for page in pages: | |
| if not isinstance(page, dict): | |
| raise ValueError("invalid parse page") | |
| page_type = page.get("type") | |
| if page_type == "blocks": | |
| content, page_boxes = _shape_blocks_page(page) | |
| elif page_type == "markdown": | |
| content, page_boxes = _shape_markdown_page(page) | |
| elif page_type == "raw_generation" and isinstance( | |
| page.get("raw_generation"), str | |
| ): | |
| content, page_boxes = page["raw_generation"], [] | |
| else: | |
| raise ValueError("unsupported parse page") | |
| page_content.append(content) | |
| boxes.extend(page_boxes) | |
| text_output = "\n\n".join(page_content) | |
| shaped = shape_parser_output(text_output) | |
| return { | |
| **shaped, | |
| "raw_output": json.dumps(body, indent=2, ensure_ascii=False), | |
| "output_chars": len(text_output), | |
| "boxes": boxes[:250], | |
| "text_output": text_output, | |
| "response": body, | |
| } | |