AI News HubLIVE
サイト内リライト6 分で読了

翻訳待ち:Pixel-Native RAG: A Practical Guide to Visual Document Indexing

AI サービスが一時的に利用できないため、復旧後に翻訳を補完します。ソース概要:Move beyond traditional text-based parsing with PixelRAG, an end-to-end system that treats web pages and PDFs as images. This tutorial explores the complete pipeline—from rendering and tiling to multimodal embedding and hybrid search—enabling developers to build high-performance, visual document retrieval systems The post Pixel-Native RAG: A Practical Guide to Visual Document Indexing appeared first on MarkTechPost.

ソースMarkTechPost著者: Sana Hassan

AI サービスが一時的に利用できないため、復旧後に翻訳を補完します。

In this tutorial, we build a complete pixel-native retrieval-augmented generation pipeline from scratch and examine how document retrieval works without relying on conventional HTML parsing, text extraction, or fixed chunking strategies. We render web pages and PDF documents as images, divide them into overlapping tiles, generate multimodal embeddings with SigLIP, CLIP, or an optional Qwen3-VL backend, and store the resulting vectors in a FAISS index for efficient similarity search. We also strengthen retrieval with OCR-based BM25 scoring and reciprocal rank fusion, aggregate tile-level evidence into document-level results, and expose the system through a FastAPI search service. Along the way, we evaluate retrieval quality using Recall@k and mean reciprocal rank, train a lightweight residual adapter with contrastive learning, visualize retrieved screenshots, and optionally pass the strongest evidence tiles to a vision-language model for grounded answer generation. Copy CodeCopiedUse a different Browser import os import sys import io import re import json import time import math import shutil import hashlib import asyncio import logging import argparse import threading import subprocess from pathlib import Path from dataclasses import dataclass, field, asdict from typing import List, Dict, Any, Optional, Tuple @dataclass class Config: urls: List[str] = field(default_factory=lambda: [ "https://en.wikipedia.org/wiki/Retrieval-augmented_generation", "https://en.wikipedia.org/wiki/Vector_database", "https://en.wikipedia.org/wiki/Transformer_(deep_learning_architecture)", "https://en.wikipedia.org/wiki/Photosynthesis", "https://en.wikipedia.org/wiki/Delhi", ]) include_synthetic_pdf: bool = True tile_width: int = 1024 tile_height: int = 1024 tile_overlap: int = 128 device_scale: float = 1.0 max_page_height: int = 24000 max_tiles_per_doc: int = 12 min_tile_height: int = 200 blank_std_threshold: float = 6.0 dedup_hamming: int = 4 nav_timeout_ms: int = 60000 headless_args: List[str] = field(default_factory=lambda: [ "--no-sandbox", "--disable-dev-shm-usage", "--hide-scrollbars", "--disable-gpu", "--force-color-profile=srgb", "--font-render-hinting=none", ]) backend: str = "siglip" model_id: str = "google/siglip-base-patch16-224" qwen_model_id: str = "Qwen/Qwen3-VL-Embedding-2B" embed_batch_size: int = 8 embed_image_size: Optional[int] = None index_dir: str = "./pixel_index" ivf_threshold: int = 2000 ivf_nprobe: int = 16 top_k_tiles: int = 20 n_docs: int = 5 use_ocr_hybrid: bool = True rrf_k: int = 60 dense_weight: float = 1.0 sparse_weight: float = 1.0 enable_server: bool = True server_port: int = 8000 enable_eval: bool = True enable_adapter_train: bool = True enable_vlm_answer: bool = False vlm_model_id: str = "Qwen/Qwen2.5-VL-3B-Instruct" show_plots: bool = True work_dir: str = "./pixelrag_work" seed: int = 0 CFG = Config() EVAL_QUERIES: List[Tuple[str, str]] = [ ("how do plants convert sunlight into chemical energy", "Photosynthesis"), ("chlorophyll light dependent reactions", "Photosynthesis"), ("converting scanned images of text into machine readable characters", "Optical_character"), ("approximate nearest neighbour search over embeddings", "Vector_database"), ("self-attention multi-head architecture", "Transformer"), ("grounding a language model with retrieved documents", "Retrieval-augmented"), ("capital territory of india red fort", "Delhi"), ] logging.basicConfig(level=logging.INFO, format="%(asctime)s | %(levelname)-7s | %(message)s", datefmt="%H:%M:%S") log = logging.getLogger("pixelrag") for noisy in ("urllib3", "PIL", "matplotlib", "httpx", "asyncio", "uvicorn.error"): logging.getLogger(noisy).setLevel(logging.WARNING) IN_COLAB = "google.colab" in sys.modules def _pip(*pkgs: str) -> None: """Install quietly; never explode the notebook on a single bad wheel.""" cmd = [sys.executable, "-m", "pip", "install", "-q", "--disable-pip-version-check", *pkgs] subprocess.run(cmd, check=False, stdout=subprocess.DEVNULL, stderr=subprocess.STDOUT) def _have(mod: str) -> bool: import importlib.util return importlib.util.find_spec(mod) is not None def ensure_deps(cfg: Config) -> None: log.info("Installing dependencies (first run only, ~2-4 min)...") wanted = [] for mod, pkg in [ ("PIL", "pillow"), ("numpy", "numpy"), ("faiss", "faiss-cpu"), ("fitz", "pymupdf"), ("transformers", "transformers"), ("fastapi", "fastapi"), ("uvicorn", "uvicorn"), ("requests", "requests"), ("matplotlib", "matplotlib"), ("tqdm", "tqdm"), ("rank_bm25", "rank-bm25"), ("playwright", "playwright"), ("sentencepiece", "sentencepiece"), ]: if not _have(mod): wanted.append(pkg) if cfg.use_ocr_hybrid and not _have("pytesseract"): wanted.append("pytesseract") if wanted: _pip(*wanted) if not _have("torch"): log.warning("torch not found — installing CPU wheel (Colab normally ships torch).") _pip("torch", "torchvision") if cfg.use_ocr_hybrid and shutil.which("tesseract") is None: log.info("Installing tesseract-ocr system package...") subprocess.run("apt-get -qq update && apt-get -qq install -y tesseract-ocr", shell=True, check=False, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL) if shutil.which("tesseract") is None: log.warning("tesseract unavailable -> hybrid retrieval will run dense-only.") cfg.use_ocr_hybrid = False marker = Path(cfg.work_dir) / ".chromium_ok" if not marker.exists(): log.info("Downloading Playwright Chromium...") r = subprocess.run([sys.executable, "-m", "playwright", "install", "--with-deps", "chromium"], capture_output=True, text=True) if r.returncode != 0: r = subprocess.run([sys.executable, "-m", "playwright", "install", "chromium"], capture_output=True, text=True) if r.returncode == 0: marker.parent.mkdir(parents=True, exist_ok=True) marker.write_text("ok") else: log.warning("Chromium install failed -> falling back to the text renderer.\n%s", (r.stderr or "")[-600:]) log.info("Dependencies ready.") def run_async(coro): """ Run a coroutine from a Jupyter/Colab cell. Colab already owns a running event loop, which makes Playwright's *sync* API raise. Rather than monkey-patching with nest_asyncio, we hand the coroutine to a private loop on a private thread — the most robust option. """ box: Dict[str, Any] = {} def _runner(): loop = asyncio.new_event_loop() asyncio.set_event_loop(loop) try: box["value"] = loop.run_until_complete(coro) except BaseException as exc: box["error"] = exc finally: try: loop.run_until_complete(loop.shutdown_asyncgens()) finally: loop.close() t = threading.Thread(target=_runner, daemon=True) t.start() t.join() if "error" in box: raise box["error"] return box.get("value") We define the global configuration, evaluation queries, logging behavior, and runtime settings for the PixelRAG pipeline. We install the required Python and system dependencies, including Playwright, Chromium, Tesseract, FAISS, and transformer libraries. We also create an asynchronous execution helper that allows browser-rendering coroutines to run reliably inside Google Colab and Jupyter environments. Copy CodeCopiedUse a different Browser @dataclass class Tile: tile_id: str doc_id: str source: str kind: str page: int seq: int y0: int y1: int path: str ocr_text: str = "" title: str = "" def _doc_id_from_source(src: str) -> str: tail = src.rstrip("/").split("/")[-1] or src tail = re.sub(r"\.(html?|pdf|png|jpg)$", "", tail, flags=re.I) return re.sub(r"[^A-Za-z0-9_.\-()]+", "_", tail)[:80] or hashlib.md5(src.encode()).hexdigest()[:10] def _ahash(img, size: int = 8) -> int: """64-bit average hash — cheap near-duplicate detection for repeated headers.""" import numpy as np g = img.convert("L").resize((size, size)) a = np.asarray(g, dtype="float32") bits = (a > a.mean()).flatten() out = 0 for b in bits: out = (out int: return bin(a ^ b).count("1") def _is_informative(img, cfg: Config) -> bool: """Reject blank / solid-colour tiles before they ever reach the GPU.""" import numpy as np a = np.asarray(img.convert("L"), dtype="float32") return float(a.std()) >= cfg.blank_std_threshold def _save_tile(img, out_dir: Path, name: str) -> str: out_dir.mkdir(parents=True, exist_ok=True) p = out_dir / f"{name}.png" img.convert("RGB").save(p, format="PNG", optimize=True) return str(p) def slice_image_to_tiles(img, cfg: Config, *, doc_id: str, source: str, kind: str, page: int, out_dir: Path, start_seq: int = 0, seen_hashes: Optional[List[int]] = None, title: str = "") -> List[Tile]: """Vertical sliding window with overlap. Used for PDFs and text fallback.""" from PIL import Image seen_hashes = seen_hashes if seen_hashes is not None else [] W, H = img.size if W != cfg.tile_width: new_h = max(1, int(H * cfg.tile_width / W)) img = img.resize((cfg.tile_width, new_h)) W, H = img.size step = max(1, cfg.tile_height - cfg.tile_overlap) tiles: List[Tile] = [] y, seq = 0, start_seq while y start_seq: break crop = img.crop((0, y, W, y + h)) if _is_informative(crop, cfg): hsh = _ahash(crop) if all(_hamming(hsh, s) > cfg.dedup_hamming for s in seen_hashes): seen_hashes.append(hsh) tid = f"{doc_id}p{page}t{seq}" tiles.append(Tile( tile_id=tid, doc_id=doc_id, source=source, kind=kind, page=page, seq=seq, y0=y, y1=y + h, title=title, path=_save_tile(crop, out_dir, tid), )) seq += 1 y += step return tiles _JS_AUTOSCROLL = """ async () => { await new Promise((resolve) => { let y = 0; const timer = setInterval(() => { window.scrollBy(0, 800); y += 800; if (y >= document.body.scrollHeight || y > 40000) { clearInterval(timer); window.scrollTo(0, 0); setTimeout(resolve, 250); } }, 40); }); } """ _JS_FLATTEN = """ () => { document.querySelectorAll('*').forEach((el) => { const s = getComputedStyle(el); if (s.position === 'fixed' || s.position === 'sticky') el.style.position = 'absolute'; }); document.querySelectorAll('[role="dialog"], .cookie, #cookie-banner, .cc-banner') .forEach((el) => el.remove()); } """ _CSS_CLEANUP = """ * { animation: none !important; transition: none !important; scroll-behavior: auto !important; } html { -webkit-font-smoothing: antialiased; } video, iframe[src*="youtube"] { visibility: hidden !important; } """ _UA = ("Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) " "Chrome/124.0 Safari/537.36 PixelRAG-Tutorial/1.0") async def _render_urls_async(urls: List[str], cfg: Config, out_dir: Path) -> List[Tile]: from playwright.async_api import async_playwright from PIL import Image all_tiles: List[Tile] = [] async with async_playwright() as pw: browser = await pw.chromium.launch(headless=True, args=cfg.headless_args) ctx = await browser.new_context( viewport={"width": cfg.tile_width, "height": cfg.tile_height}, device_scale_factor=cfg.device_scale, user_agent=_UA, java_script_enabled=True, ) for url in urls: doc_id = _doc_id_from_source(url) page = await ctx.new_page() try: await page.goto(url, wait_until="domcontentloaded", timeout=cfg.nav_timeout_ms) try: await page.wait_for_load_state("networkidle", timeout=12000) except Exception: pass await page.evaluate(_JS_AUTOSCROLL) await page.add_style_tag(content=_CSS_CLEANUP) await page.evaluate(_JS_FLATTEN) title = (await page.title()) or doc_id height = await page.evaluate( "() => Math.max(document.body.scrollHeight, " "document.documentElement.scrollHeight)") height = int(min(height, cfg.max_page_height)) step = max(1, cfg.tile_height - cfg.tile_overlap) seen: List[int] = [] y, seq = 0, 0 while y 0: break buf = await page.screenshot( full_page=True, type="png", clip={"x": 0, "y": y, "width": cfg.tile_width, "height": h}) img = Image.open(io.BytesIO(buf)).convert("RGB") if img.size[0] != cfg.tile_width: img = img.resize((cfg.tile_width, max(1, int(img.size[1] * cfg.tile_width / img.size[0])))) if _is_informative(img, cfg): hsh = _ahash(img) if all(_hamming(hsh, s) > cfg.dedup_hamming for s in seen): seen.append(hsh) tid = f"{doc_id}p0t{seq}" all_tiles.append(Tile( tile_id=tid, doc_id=doc_id, source=url, kind="web", page=0, seq=seq, y0=y, y1=y + h, title=title, path=_save_tile(img, out_dir, tid))) seq += 1 y + [truncated for AI cost control]