# -*- coding: utf-8 -*-
"""
Builds the seven files of the public input set `documents-v1` - the set every document tool on
MarketAIVerse is measured against.

The files are never written or edited by hand. They are regenerated by this script, so they are
identical for everyone and every time. Run it and compare the checksums with the files we
publish; if they differ, tell us.

Why the text inside the PDFs is in Romanian: the set tests what document tools most often get
wrong on real manuals, and one of those things is Romanian diacritics (a-breve, a-circumflex,
i-circumflex, s-comma, t-comma - and the classic trap of s/t with a CEDILLA instead of a comma).
The labels, tables and paragraphs are therefore Romanian on purpose. They are the test data,
not a translation we forgot.

Run:     python make_set.py
Output:  ./documents-v1/
Needs:   PyMuPDF (pip install pymupdf), and a font that has the Romanian letters (Arial, Tahoma,
         Segoe UI, Calibri, DejaVu Sans or Liberation Sans). File 4 depends on the font, so it
         is byte-identical to ours only with the same font file.
"""

import hashlib
import os
import re
import fitz  # PyMuPDF

HERE = os.path.dirname(os.path.abspath(__file__))
OUT = os.path.join(HERE, "documents-v1")
os.makedirs(OUT, exist_ok=True)

BLACK = (0.05, 0.13, 0.11)
GREY = (0.35, 0.40, 0.38)

# The set was first published under Romanian file names. Each PDF's internal /ID is derived
# from that ORIGINAL name, so the bytes stay identical to the files we measured - renaming a
# file must not change what is inside it.
ORIGINAL_NAME = {
    "1-simple.pdf": "1-simplu.pdf", "2-tables.pdf": "2-tabele.pdf",
    "3-text-in-images.pdf": "3-text-in-imagini.pdf", "4-diacritics.pdf": "4-diacritice.pdf",
    "5-large.pdf": "5-mare.pdf", "6-hollowed.pdf": "6-golit.pdf",
    "7-no-header.pdf": "7-fara-antet.pdf",
}


def new_page(doc, title):
    p = doc.new_page(width=595, height=842)  # A4
    p.insert_text((56, 70), title, fontsize=17, fontname="hebo", color=BLACK)
    p.draw_line(fitz.Point(56, 82), fitz.Point(539, 82), color=GREY, width=0.7)
    return p


# The built-in `helv` font has no Romanian letters: PyMuPDF silently replaces them with a dot.
# We found out the hard way - file 4, the diacritics test, once contained no diacritics at all.
# So file 4 uses a real font, and the script reads the file back at the end to prove it.
FONTS_WITH_DIACRITICS = [
    "C:/Windows/Fonts/arial.ttf", "C:/Windows/Fonts/tahoma.ttf",
    "C:/Windows/Fonts/segoeui.ttf", "C:/Windows/Fonts/calibri.ttf",
    "/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf",
    "/usr/share/fonts/truetype/liberation/LiberationSans-Regular.ttf",
]
REQUIRED = "ăâîșțĂÂÎȘȚ"


def good_font():
    """A font that really has the Romanian letters. Checked, not assumed."""
    for path in FONTS_WITH_DIACRITICS:
        if not os.path.exists(path):
            continue
        try:
            f = fitz.Font(fontfile=path)
            if all(f.has_glyph(ord(c)) for c in REQUIRED):
                return path
        except Exception:
            continue
    raise SystemExit("No font with the Romanian letters found. Without one, file 4 tests "
                     "nothing - add a font path to FONTS_WITH_DIACRITICS.")


def fix_id(path):
    """A PDF gets a RANDOM identifier on every save, so two runs would differ by ~55 bytes.
    We derive it from the file's original name instead: stable across runs, different per file."""
    data = open(path, "rb").read()
    mark = hashlib.md5(ORIGINAL_NAME[os.path.basename(path)].encode("utf-8")).hexdigest().upper()
    new = b"/ID[<" + mark.encode() + b"><" + mark.encode() + b">]"
    changed, n = re.subn(rb"/ID\s*\[\s*<[0-9A-Fa-f]*>\s*<[0-9A-Fa-f]*>\s*\]", new, data)
    if n:
        open(path, "wb").write(changed)
    return n


def write(p, y, text, size=10.5, font="helv", colour=BLACK):
    p.insert_text((56, y), text, fontsize=size, fontname=font, color=colour)
    return y + size + 5


# ---------------------------------------------------------------- 1. a plain document
def one():
    doc = fitz.open()
    for i in range(1, 6):
        p = new_page(doc, "Sectiunea %d. Instructiuni de utilizare" % i)
        y = 120
        for r in range(18):
            y = write(p, y, "Rand %d. Text simplu, fara nimic special, doar ca sa existe volum "
                            "si sa se poata numara ce a intrat si ce a iesit." % (r + 1))
        p.insert_text((56, 800), "pagina %d din 5" % i, fontsize=8, color=GREY)
    path = os.path.join(OUT, "1-simple.pdf")
    doc.save(path)
    fix_id(path)
    doc.close()
    return "1-simple.pdf", "plain document, clean text, 5 pages", "must come out perfect"


# ---------------------------------------------------------------- 2. tables across pages
def two():
    doc = fitz.open()
    rows = [("Cod", "Denumire", "Tensiune", "Curent", "Masa"),
            ("A-101", "Motor principal", "230 V", "4,2 A", "3,10 kg"),
            ("A-102", "Motor auxiliar", "230 V", "1,8 A", "1,45 kg"),
            ("B-201", "Placa de comanda", "24 V", "0,6 A", "0,22 kg"),
            ("B-202", "Senzor de temperatura", "5 V", "0,02 A", "0,04 kg"),
            ("C-310", "Cablu de alimentare", "-", "-", "0,80 kg")]
    widths = [70, 190, 80, 70, 80]
    for page in range(1, 4):
        p = new_page(doc, "Tabelul %d. Date tehnice" % page)
        y = 120
        for idx in range(14):  # a table that flows over several pages
            r = rows[idx % len(rows)]
            x = 56
            bold = "hebo" if idx % len(rows) == 0 else "helv"
            for cell, w in zip(r, widths):
                p.insert_text((x + 3, y), cell, fontsize=9.5, fontname=bold, color=BLACK)
                x += w
            p.draw_line(fitz.Point(56, y + 4), fitz.Point(56 + sum(widths), y + 4), color=GREY, width=0.35)
            y += 20
        x = 56
        for w in widths + [0]:
            p.draw_line(fitz.Point(x, 110), fitz.Point(x, y - 16), color=GREY, width=0.35)
            x += w
    path = os.path.join(OUT, "2-tables.pdf")
    doc.save(path)
    fix_id(path)
    doc.close()
    return "2-tables.pdf", "tables across several pages", "tables intact, not broken apart"


# ---------------------------------------------------------------- 3. text inside images
def three():
    """The text is DRAWN, not written as text - it cannot be extracted, it has to be read from
    the image. This is the trap that leaves figures untranslated almost everywhere."""
    doc = fitz.open()
    for page in range(1, 3):
        p = new_page(doc, "Figura %d. Schema de montaj" % page)
        p.draw_rect(fitz.Rect(90, 140, 300, 300), color=BLACK, width=1.2)
        p.draw_rect(fitz.Rect(340, 140, 500, 240), color=BLACK, width=1.2)
        p.draw_line(fitz.Point(300, 200), fitz.Point(340, 200), color=BLACK, width=1.2)
        p.draw_circle(fitz.Point(195, 360), 34, color=BLACK, width=1.2)
        for point, label in (((110, 165), "CARCASA PRINCIPALA"), ((355, 165), "MODUL DE COMANDA"),
                             ((150, 420), "SURUB M6 (4 buc.)"), ((110, 275), "ATENTIE: 230 V")):
            p.insert_text(point, label, fontsize=9, fontname="hebo", color=BLACK, render_mode=0)
        p.insert_text((56, 520), "Textul de mai sus face parte din figura. O traducere care nu se "
                                 "uita in imagine il lasa neatins.", fontsize=9.5, color=GREY)
    # Every page becomes a picture, so no text layer is left at all.
    # JPEG at 110 dpi, not PNG at 150: PNG made the file 12.7 MB, far too big to download.
    images = fitz.open()
    for p in doc:
        pix = p.get_pixmap(dpi=110)
        jpg = pix.tobytes("jpeg", jpg_quality=78)
        np = images.new_page(width=595, height=842)
        np.insert_image(fitz.Rect(0, 0, 595, 842), stream=jpg)
    path = os.path.join(OUT, "3-text-in-images.pdf")
    images.save(path, deflate=True)
    fix_id(path)
    images.close()
    doc.close()
    return "3-text-in-images.pdf", "text drawn into figures, no text layer", "the text in the drawings handled"


# ---------------------------------------------------------------- 4. diacritics
def four():
    doc = fitz.open()
    p = new_page(doc, "Verificare de caractere")
    y = 120
    lines = [
        "Romana:  a a i s t   A A I S T   (a-caciula, a-din-a, i-din-i, s-virgula, t-virgula)",
        "Reale:   ă â î ș ț   Ă Â Î Ș Ț",
        "Gresite: ş ţ Ş Ţ   (s si t cu sedila, NU cu virgula - capcana clasica)",
        "Germana: ä ö ü ß      Franceza: é è ê ç à",
        "Semne:   °C  ±  ×  –  —  „”  «»  µm  Ω  €",
        "Cifre:   0,5 kg   1.200 buc   3,14   10³   45°",
    ]
    font_path = good_font()
    for l in lines:
        p.insert_text((56, y), l, fontsize=11, fontname="rom", fontfile=font_path, color=BLACK)
        y += 11 * 1.45 + 6
    y += 14
    y = write(p, y, "Daca vreunul din caracterele de mai sus lipseste sau se schimba in altceva,", 10, colour=GREY)
    y = write(p, y, "proba a picat la punctul 4. Nu e o chestie de finete: un manual in romana", 10, colour=GREY)
    write(p, y, "fara diacritice arata a facut in graba.", 10, colour=GREY)
    path = os.path.join(OUT, "4-diacritics.pdf")
    # The font is embedded as a SUBSET - only the glyphs used. Without it the file jumps from
    # about 50 KB to 1 MB for six lines of text.
    doc.subset_fonts()
    doc.save(path, garbage=4, deflate=True, clean=True)
    fix_id(path)
    doc.close()
    # Read back what was written. This is the step that was missing when file 4 had no diacritics.
    d = fitz.open(path)
    out = "".join(pg.get_text() for pg in d)
    d.close()
    missing = [c for c in REQUIRED if c not in out]
    if missing:
        raise SystemExit("4-diacritics.pdf does NOT contain: %s - the font did not write them. "
                         "Do not use this file." % "".join(missing))
    return "4-diacritics.pdf", "diacritics and special characters", "no character lost"


# ---------------------------------------------------------------- 5. a large document
def five():
    doc = fitz.open()
    sections = ["Introducere", "Continutul pachetului", "Date tehnice", "Montaj",
                "Punere in functiune", "Utilizare zilnica", "Intretinere", "Depanare",
                "Piese de schimb", "GARANTIE", "Conformitate", "Contact"]
    n_fig = 0
    for i, s in enumerate(sections):
        for sub in range(5):
            p = new_page(doc, "%d.%d %s" % (i + 1, sub + 1, s))
            y = 120
            for r in range(14):
                y = write(p, y, "Paragraf %d din sectiunea %s. Text de umplutura cu volum realist." % (r + 1, s))
            if sub % 2 == 0:
                n_fig += 1
                p.draw_rect(fitz.Rect(56, y + 10, 300, y + 130), color=GREY, width=0.8)
                p.insert_text((62, y + 145), "Figura %d" % n_fig, fontsize=9, fontname="hebo", color=BLACK)
            p.insert_text((56, 800), "pagina %d" % (len(doc)), fontsize=8, color=GREY)
    path = os.path.join(OUT, "5-large.pdf")
    doc.save(path)
    fix_id(path)
    n = len(doc)
    doc.close()
    return "5-large.pdf", "large document: %d pages, %d figures" % (n, n_fig), "finishes, no section lost"


# ------------------------------------------------- 6 and 7. THE CASES THAT MUST FAIL
#
# Measured on 13 Sep 2026, and it changed how the set is built: a PDF reader almost NEVER
# refuses. It repairs and returns something.
#   cut to 55%  -> opens, 6330 of 6345 characters (99.8%)
#   cut to 30%  -> opens, 52% of the text
#   cut to  8%  -> opens, 9.6% of the text
#   cut to  4%  -> opens, reports "5 pages", holds 39 characters
#   bytes shuffled -> opens, 8460 characters: GARBAGE, longer than the original
#   header destroyed -> 0 pages
# So the right question is not "does it refuse?" but "does it notice content is missing?".

def six():
    """Opens, says it has 5 pages, but under 1% of the text is left.
    The tool must NOTICE the content is missing. Reporting success = failing the test."""
    data = open(os.path.join(OUT, "1-simple.pdf"), "rb").read()
    cut = data[: int(len(data) * 0.04)]
    open(os.path.join(OUT, "6-hollowed.pdf"), "wb").write(cut)
    return ("6-hollowed.pdf",
            "cut to 4%% (%d of %d bytes) - OPENS, but is empty inside" % (len(cut), len(data)),
            "must NOTICE the missing content. Reported success = failed")


def seven():
    """The header is destroyed: nothing left to recover, zero pages."""
    data = open(os.path.join(OUT, "1-simple.pdf"), "rb").read()
    open(os.path.join(OUT, "7-no-header.pdf"), "wb").write(b"%PDF-1.7\n" + data[200:])
    return "7-no-header.pdf", "header replaced: zero recoverable pages", "MUST REFUSE clearly"


if __name__ == "__main__":
    results = [one(), two(), three(), four(), five(), six(), seven()]
    print("Set documents-v1, in %s\n" % OUT)
    for i, (name, what, expected) in enumerate(results, 1):
        k = os.path.getsize(os.path.join(OUT, name)) / 1024.0
        h = hashlib.sha256(open(os.path.join(OUT, name), "rb").read()).hexdigest()[:16]
        print("  %d. %-22s %7.1f KB  sha256 %s  %s" % (i, name, k, h, what))
        print("     expected: %s" % expected)
