Close Menu
    Facebook X (Twitter) Instagram
    • Privacy Policy
    • Terms Of Service
    • Social Media Disclaimer
    • DMCA Compliance
    • Anti-Spam Policy
    Facebook X (Twitter) Instagram
    Deep Tech Ledger
    • Home
    • Crypto News
      • Bitcoin
      • Ethereum
      • Altcoins
      • Blockchain
      • DeFi
    • AI News
    • Stock News
    • Learn
      • AI for Beginners
      • AI Tips
      • Make Money with AI
    • Reviews
    • Tools
      • Best AI Tools
      • Crypto Market Cap List
      • Stock Market Overview
      • Market Heatmap
    • Contact
    Deep Tech Ledger
    Home»AI News»Pixel-Native RAG: A Practical Guide to Visual Document Indexing
    Pixel-Native RAG: A Practical Guide to Visual Document Indexing
    AI News

    Pixel-Native RAG: A Practical Guide to Visual Document Indexing

    August 5, 20267 Mins Read
    Share
    Facebook Twitter LinkedIn Pinterest Email
    murf


    @dataclass
    class Tile:
    tile_id: str
    doc_id: str
    source: str
    kind: str
    page: int
    seq: int
    y0: int
    y1: int
    path: str
    ocr_text: str = “”
    title: str = “”
    def _doc_id_from_source(src: str) -> str:
    tail = src.rstrip(“/”).split(“/”)[-1] or src
    tail = re.sub(r”\.(html?|pdf|png|jpg)$”, “”, tail, flags=re.I)
    return re.sub(r”[^A-Za-z0-9_.\-()]+”, “_”, tail)[:80] or hashlib.md5(src.encode()).hexdigest()[:10]
    def _ahash(img, size: int = 8) -> int:
    “””64-bit average hash — cheap near-duplicate detection for repeated headers.”””
    import numpy as np
    g = img.convert(“L”).resize((size, size))
    a = np.asarray(g, dtype=”float32″)
    bits = (a > a.mean()).flatten()
    out = 0
    for b in bits:
    out = (out << 1) | int(b)
    return out
    def _hamming(a: int, b: int) -> int:
    return bin(a ^ b).count(“1”)
    def _is_informative(img, cfg: Config) -> bool:
    “””Reject blank / solid-colour tiles before they ever reach the GPU.”””
    import numpy as np
    a = np.asarray(img.convert(“L”), dtype=”float32″)
    return float(a.std()) >= cfg.blank_std_threshold
    def _save_tile(img, out_dir: Path, name: str) -> str:
    out_dir.mkdir(parents=True, exist_ok=True)
    p = out_dir / f”{name}.png”
    img.convert(“RGB”).save(p, format=”PNG”, optimize=True)
    return str(p)
    def slice_image_to_tiles(img, cfg: Config, *, doc_id: str, source: str, kind: str,
    page: int, out_dir: Path, start_seq: int = 0,
    seen_hashes: Optional[List[int]] = None,
    title: str = “”) -> List[Tile]:
    “””Vertical sliding window with overlap. Used for PDFs and text fallback.”””
    from PIL import Image
    seen_hashes = seen_hashes if seen_hashes is not None else []
    W, H = img.size
    if W != cfg.tile_width:
    new_h = max(1, int(H * cfg.tile_width / W))
    img = img.resize((cfg.tile_width, new_h))
    W, H = img.size
    step = max(1, cfg.tile_height – cfg.tile_overlap)
    tiles: List[Tile] = []
    y, seq = 0, start_seq
    while y < H and (seq – start_seq) < cfg.max_tiles_per_doc:
    h = min(cfg.tile_height, H – y)
    if h < cfg.min_tile_height and seq > start_seq:
    break
    crop = img.crop((0, y, W, y + h))
    if _is_informative(crop, cfg):
    hsh = _ahash(crop)
    if all(_hamming(hsh, s) > cfg.dedup_hamming for s in seen_hashes):
    seen_hashes.append(hsh)
    tid = f”{doc_id}__p{page}__t{seq}”
    tiles.append(Tile(
    tile_id=tid, doc_id=doc_id, source=source, kind=kind, page=page,
    seq=seq, y0=y, y1=y + h, title=title,
    path=_save_tile(crop, out_dir, tid),
    ))
    seq += 1
    y += step
    return tiles
    _JS_AUTOSCROLL = “””
    async () => {
    await new Promise((resolve) => {
    let y = 0;
    const timer = setInterval(() => {
    window.scrollBy(0, 800);
    y += 800;
    if (y >= document.body.scrollHeight || y > 40000) {
    clearInterval(timer);
    window.scrollTo(0, 0);
    setTimeout(resolve, 250);
    }
    }, 40);
    });
    }
    “””
    _JS_FLATTEN = “””
    () => {
    document.querySelectorAll(‘*’).forEach((el) => {
    const s = getComputedStyle(el);
    if (s.position === ‘fixed’ || s.position === ‘sticky’) el.style.position = ‘absolute’;
    });
    document.querySelectorAll(‘[role=”dialog”], .cookie, #cookie-banner, .cc-banner’)
    .forEach((el) => el.remove());
    }
    “””
    _CSS_CLEANUP = “””
    * { animation: none !important; transition: none !important;
    scroll-behavior: auto !important; }
    html { -webkit-font-smoothing: antialiased; }
    video, iframe[src*=”youtube”] { visibility: hidden !important; }
    “””
    _UA = (“Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) ”
    “Chrome/124.0 Safari/537.36 PixelRAG-Tutorial/1.0”)
    async def _render_urls_async(urls: List[str], cfg: Config, out_dir: Path) -> List[Tile]:
    from playwright.async_api import async_playwright
    from PIL import Image
    all_tiles: List[Tile] = []
    async with async_playwright() as pw:
    browser = await pw.chromium.launch(headless=True, args=cfg.headless_args)
    ctx = await browser.new_context(
    viewport={“width”: cfg.tile_width, “height”: cfg.tile_height},
    device_scale_factor=cfg.device_scale,
    user_agent=_UA,
    java_script_enabled=True,
    )
    for url in urls:
    doc_id = _doc_id_from_source(url)
    page = await ctx.new_page()
    try:
    await page.goto(url, wait_until=”domcontentloaded”, timeout=cfg.nav_timeout_ms)
    try:
    await page.wait_for_load_state(“networkidle”, timeout=12000)
    except Exception:
    pass
    await page.evaluate(_JS_AUTOSCROLL)
    await page.add_style_tag(content=_CSS_CLEANUP)
    await page.evaluate(_JS_FLATTEN)
    title = (await page.title()) or doc_id
    height = await page.evaluate(
    “() => Math.max(document.body.scrollHeight, ”
    “document.documentElement.scrollHeight)”)
    height = int(min(height, cfg.max_page_height))
    step = max(1, cfg.tile_height – cfg.tile_overlap)
    seen: List[int] = []
    y, seq = 0, 0
    while y < height and seq < cfg.max_tiles_per_doc:
    h = min(cfg.tile_height, height – y)
    if h < cfg.min_tile_height and seq > 0:
    break
    buf = await page.screenshot(
    full_page=True, type=”png”,
    clip={“x”: 0, “y”: y, “width”: cfg.tile_width, “height”: h})
    img = Image.open(io.BytesIO(buf)).convert(“RGB”)
    if img.size[0] != cfg.tile_width:
    img = img.resize((cfg.tile_width,
    max(1, int(img.size[1] * cfg.tile_width / img.size[0]))))
    if _is_informative(img, cfg):
    hsh = _ahash(img)
    if all(_hamming(hsh, s) > cfg.dedup_hamming for s in seen):
    seen.append(hsh)
    tid = f”{doc_id}__p0__t{seq}”
    all_tiles.append(Tile(
    tile_id=tid, doc_id=doc_id, source=url, kind=”web”,
    page=0, seq=seq, y0=y, y1=y + h, title=title,
    path=_save_tile(img, out_dir, tid)))
    seq += 1
    y += step
    log.info(” rendered %-34s -> %2d tiles (page %dpx)”, doc_id, seq, height)
    except Exception as exc:
    log.warning(” FAILED %s (%s)”, url, type(exc).__name__)
    finally:
    await page.close()
    await ctx.close()
    await browser.close()
    return all_tiles
    def render_urls(urls: List[str], cfg: Config, out_dir: Path) -> List[Tile]:
    “””Screenshot every URL into tiles; degrade to the text renderer on failure.”””
    try:
    tiles = run_async(_render_urls_async(urls, cfg, out_dir))
    if tiles:
    return tiles
    log.warning(“Browser produced no tiles — using text-render fallback.”)
    except Exception as exc:
    log.warning(“Playwright unavailable (%s: %s) — using text-render fallback.”,
    type(exc).__name__, str(exc)[:160])
    return [t for u in urls for t in render_url_as_text(u, cfg, out_dir)]
    def _strip_html(html: str) -> str:
    html = re.sub(r”(?is)<(script|style|nav|footer|header|noscript).*?</\1>”, ” “, html)
    html = re.sub(r”(?s)<!–.*?–>”, ” “, html)
    html = re.sub(r”(?i)</(p|div|h[1-6]|li|tr|br)>”, “\n”, html)
    text = re.sub(r”(?s)<[^>]+>”, ” “, html)
    for a, b in [(” “, ” “), (“&”, “&”), (“<“, “<“), (“>”, “>”), (“””, ‘”‘)]:
    text = text.replace(a, b)
    text = re.sub(r”\[\d+\]”, “”, text)
    text = re.sub(r”[ \t]+”, ” “, text)
    return re.sub(r”\n{2,}”, “\n”, text).strip()
    def _mono_font(size: int = 20):
    from PIL import ImageFont
    for cand in (“/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf”,
    “/usr/share/fonts/truetype/liberation/LiberationSans-Regular.ttf”):
    if os.path.exists(cand):
    return ImageFont.truetype(cand, size)
    try:
    import matplotlib.font_manager as fm
    return ImageFont.truetype(fm.findfont(“DejaVu Sans”), size)
    except Exception:
    return ImageFont.load_default()
    def text_to_image(text: str, cfg: Config, title: str = “”) -> Any:
    “””Render plain text onto a tall white canvas — a browser-free stand-in.”””
    from PIL import Image, ImageDraw
    font, tfont = _mono_font(20), _mono_font(30)
    pad, lh, wrap = 40, 30, max(20, (cfg.tile_width – 80) // 11)
    lines: List[str] = []
    for para in text.split(“\n”):
    para = para.strip()
    if not para:
    continue
    while len(para) > wrap:
    cut = para.rfind(” “, 0, wrap)
    cut = cut if cut > 0 else wrap
    lines.append(para[:cut])
    para = para[cut:].lstrip()
    lines.append(para)
    lines = lines[:900]
    height = pad * 2 + 60 + lh * len(lines)
    img = Image.new(“RGB”, (cfg.tile_width, max(cfg.tile_height, height)), “white”)
    d = ImageDraw.Draw(img)
    d.text((pad, pad), title[:60], font=tfont, fill=(15, 15, 15))
    for i, ln in enumerate(lines):
    d.text((pad, pad + 60 + i * lh), ln, font=font, fill=(35, 35, 35))
    return img
    def render_url_as_text(url: str, cfg: Config, out_dir: Path) -> List[Tile]:
    import requests
    doc_id = _doc_id_from_source(url)
    try:
    r = requests.get(url, timeout=30, headers={“User-Agent”: _UA})
    r.raise_for_status()
    body = _strip_html(r.text)
    m = re.search(r”(?is)<title>(.*?)</title>”, r.text)
    title = m.group(1).strip() if m else doc_id
    except Exception as exc:
    log.warning(” fetch failed for %s (%s)”, url, type(exc).__name__)
    return []
    img = text_to_image(body, cfg, title=title)
    log.info(” text-rendered %-30s -> canvas %dpx”, doc_id, img.size[1])
    return slice_image_to_tiles(img, cfg, doc_id=doc_id, source=url, kind=”text”,
    page=0, out_dir=out_dir, title=title)
    def render_pdf(pdf_path: str, cfg: Config, out_dir: Path, dpi: int = 150) -> List[Tile]:
    import fitz
    from PIL import Image
    doc_id = _doc_id_from_source(pdf_path)
    tiles: List[Tile] = []
    with fitz.open(pdf_path) as doc:
    title = (doc.metadata or {}).get(“title”) or doc_id
    n_pages = doc.page_count
    for pno in range(n_pages):
    pix = doc[pno].get_pixmap(dpi=dpi)
    img = Image.frombytes(“RGB”, (pix.width, pix.height), pix.samples)
    tiles += slice_image_to_tiles(img, cfg, doc_id=doc_id, source=pdf_path,
    kind=”pdf”, page=pno, out_dir=out_dir,
    title=title)
    log.info(” rendered %-34s -> %2d tiles (%d pages)”, doc_id, len(tiles), n_pages)
    return tiles
    def make_synthetic_pdf(path: Path) -> str:
    “””A tiny PDF so the tutorial always exercises the PDF path, offline or not.”””
    import fitz
    body = [
    (“PixelRAG Internal Note”, 22),
    (“”, 12),
    (“Why pixel-native retrieval?”, 16),
    (“Parsers are per-site glue code. A renderer is one code path for every”, 11),
    (“document type: HTML, PDF, scanned fax, spreadsheet export, dashboard.”, 11),
    (“”, 11),
    (“Tiling policy”, 16),
    (“Tiles are 1024×1024 with 128px of vertical overlap. Overlap keeps a”, 11),
    (“sentence or table row from being split across two embeddings, which is”, 11),
    (“the single biggest source of recall loss in naive screenshot pipelines.”, 11),
    (“”, 11),
    (“Serving”, 16),
    (“FAISS inner-product over L2-normalised vectors equals cosine similarity.”, 11),
    (“Tile scores are max-pooled per document so one strong tile can surface”, 11),
    (“a long page, mirroring late-interaction retrieval behaviour.”, 11),
    (“”, 11),
    (“The mitochondria reference is a joke; the overlap advice is not.”, 11),
    ]
    doc = fitz.open()
    page = doc.new_page()
    y = 72
    for line, size in body:
    page.insert_text((72, y), line, fontsize=size, fontname=”helv”)
    y += size + 8
    doc.save(str(path))
    doc.close()
    return str(path)



    Source link

    aistudios
    Share. Facebook Twitter Pinterest LinkedIn Tumblr Email
    CryptoExpert
    • Website

    I’m someone who’s deeply curious about crypto and artificial intelligence. I created this site to share what I’m learning, break down complex ideas, and keep people updated on what’s happening in crypto and AI—without the unnecessary hype.

    Related Posts

    Alexander Rakhlin named director of the MIT Statistics and Data Science Center | MIT News

    August 4, 2026

    Stop graphing everything: When GraphRAG actually beats vector RAG

    August 3, 2026

    OpenAI aligns safety practices with EU AI Act’s GPAI Code

    August 2, 2026

    DeepSeek Upgrades DeepSeek-V4-Flash-0731 with Major Agentic and Coding Gains

    August 1, 2026
    Add A Comment
    Leave A Reply Cancel Reply

    coinbase
    Latest Posts

    HKMA Sets 5.00% Interest Rate for Silver Bond Payment Due 2026

    August 4, 2026

    Coldcard Exploit Tops $100M as Expert Says Stolen BTC May Be Hard to Spend

    August 4, 2026

    Alexander Rakhlin named director of the MIT Statistics and Data Science Center | MIT News

    August 4, 2026

    Ripple Invests in Zilo, Licuido in Tokenized Capital Markets Push

    August 4, 2026

    Hashdex Will Liquidate Market’s Smallest Bitcoin ETF DEFI

    August 3, 2026
    notion
    LEGAL INFORMATION
    • Privacy Policy
    • Terms Of Service
    • Social Media Disclaimer
    • DMCA Compliance
    • Anti-Spam Policy
    Top Insights

    GameStop plans $1.4 billion stock swap as Bitcoin collateral risk emerges

    August 5, 2026

    Pixel-Native RAG: A Practical Guide to Visual Document Indexing

    August 5, 2026
    kraken
    Facebook X (Twitter) Instagram Pinterest
    © 2026 DeepTechLedger.com - All rights reserved.

    Type above and press Enter to search. Press Esc to cancel.