From 4947e88032b1019fcb01c43f5d2a728eb0772656 Mon Sep 17 00:00:00 2001 From: Kehribar <103407696+dpentx@users.noreply.github.com> Date: Sun, 27 Sep 2026 21:57:11 +0300 Subject: [PATCH] =?UTF-8?q?fix:=20eptran-web=20ingest'inde=20epub=20yerine?= =?UTF-8?q?=20txt=20b=C3=B6l=C3=BCmlerini=20g=C3=B6nder?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Vercel Serverless Function istek gövdesi sınırı (~4.5MB) yüzünden 20-30MB'lık epub'lar 413 ile reddediliyordu. eptran-web'in ingest.ts'i zaten sadece düz metni saklıyor (resimleri kullanmıyor), o yüzden epub yerine küçük txt bölüm dosyaları gönderiliyor artık. --- scripts/backfill_ingest.py | 18 +++++++--- scripts/convert.py | 2 +- scripts/lib/web_ingest.py | 72 +++++++++++++++++++++----------------- 3 files changed, 54 insertions(+), 38 deletions(-) diff --git a/scripts/backfill_ingest.py b/scripts/backfill_ingest.py index 8cf4ec1d..97bcac0a 100644 --- a/scripts/backfill_ingest.py +++ b/scripts/backfill_ingest.py @@ -8,13 +8,19 @@ Bu script main'deki her tamamlanmış kitabı tek tek tarayıp push_book() ile eptran-web'e gönderir. +NOT (Eylül 2026): İlk sürüm epub gönderiyordu, Vercel'in ~4.5MB istek +sınırına tosladı (413). Artık convert.py'nin kendisiyle AYNI yolu +kullanıyor: her kitabın output//*.txt bölüm dosyalarını +load_txt_chapters() ile ayrıştırıp format=txt olarak gönderiyor — +bkz. lib/web_ingest.py'nin kendi notu. + Idempotent: aynı kitabı ikinci kez göndermek zararsız — ingest.ts aynı sourceKey'i güncelleme (upsert) olarak işliyor, kopya novel oluşmuyor. """ import os from lib.web_ingest import push_book -from convert import get_book_metadata +from convert import get_book_metadata, load_txt_chapters def find_completed_books(output_root="output"): @@ -27,7 +33,7 @@ def find_completed_books(output_root="output"): continue epub_path = os.path.join(book_dir, f"{slug}_tr.epub") if os.path.exists(epub_path): - books.append((slug, epub_path)) + books.append((slug, book_dir)) return books @@ -40,13 +46,15 @@ def main(): print(f"{len(books)} tamamlanmış kitap bulundu: {', '.join(b[0] for b in books)}") print() - for slug, epub_path in books: + for slug, book_dir in books: original_epub_path = f"input/.originals/{slug}.epub" if not os.path.exists(original_epub_path): original_epub_path = None title, author = get_book_metadata(slug, original_epub_path) - print(f"→ {slug} (\"{title}\"{f' — {author}' if author else ''})") - push_book(slug, epub_path, title, author) + chapters = load_txt_chapters(book_dir) + print(f"→ {slug} (\"{title}\"{f' — {author}' if author else ''}, " + f"{len(chapters)} bölüm)") + push_book(slug, chapters, title, author) print() diff --git a/scripts/convert.py b/scripts/convert.py index 7c6062b9..d0cb80e0 100644 --- a/scripts/convert.py +++ b/scripts/convert.py @@ -398,7 +398,7 @@ def main(): # WEB_INGEST_SECRET tanımlı değilse ya da istek başarısız olursa # sadece uyarı basar, PR açma akışını hiçbir şekilde durdurmaz. title, author = get_book_metadata(book_slug, original_epub_path) - push_book(book_slug, epub_out, title, author) + push_book(book_slug, chapters, title, author) # Kitap tamamen bitti: çeviri + review + ciltleme. Bu, tüm sürecin TEK # onay noktası — kitap dalından (book/) main'e bir PR açılıyor. diff --git a/scripts/lib/web_ingest.py b/scripts/lib/web_ingest.py index 7320cad8..f35a946e 100644 --- a/scripts/lib/web_ingest.py +++ b/scripts/lib/web_ingest.py @@ -1,23 +1,24 @@ """ eptran-web'e (Astro + Turso) tamamlanan kitapları POST /api/ingest ile -gönderir. eptran-web tarafı zaten hazır ve bekliyordu (bkz. o reponun -README'si ve src/pages/api/ingest.ts) — eksik olan taraf HER ZAMAN -buraydı: eptran'ın hiçbir scripti bu endpoint'i hiç çağırmıyordu. +gönderir. -Tasarım: convert.py, bir kitabın epub ciltlemesini bitirdiği anda -(convert_status == "completed") burayı çağırır. Bu an, README'nin -"Epub varsa her zaman epub tercih edilir" sözleşmesiyle birebir -örtüşüyor — eptran-web'in kendi parseEpub()'ı OPF spine sırasına göre -otomatik bölüyor, biz sadece dosyayı ve iki-üç metadata alanını -gönderiyoruz. +NOT (Eylül 2026, gerçek üretim hatası): İlk sürüm burada ciltlenmiş +epub'ı (resimler dahil, 20-30MB) gönderiyordu. Bu, Vercel'in Serverless +Function istek gövdesi sınırına (~4.5MB) tosladı — 413 Request Entity +Too Large. Oysa eptran-web'in kendi ingest.ts'i epub'ı parse edip SADECE +düz metni (chapters.content) veritabanına yazıyor, resimleri zaten hiç +kullanmıyor. Yani epub göndermek en başından gereksiz bir israftı. + +Artık her bölümü ayrı, küçük bir "NNN_slug.txt" dosyası olarak +(convert.py'nin load_txt_chapters()'ının zaten bellekte ayrıştırdığı +title/body ile, orijinal dosyadaki "# " başlık işaretçisi OLMADAN, +ingest.ts'in parseTxtChapter()'ının beklediği "ilk satır = başlık" +biçiminde) gönderiyoruz — tüm kitap için toplam yük genelde birkaç +yüz KB, 4.5MB sınırının çok altında. Fail-soft: internet/HTTP hatası ya da eksik secret durumunda script'i -ÇÖKERTMEZ, sadece uyarı basar — projedeki diğer "best-effort" işlemlerle -(bkz. git_utils.trigger_workflow) aynı felsefe. Kitabın kendisi (epub + -main'e PR) bu adımdan tamamen bağımsız, ingest başarısız olsa bile PR -açılmaya devam eder; bir sonraki convert.yml çalıştırması (örn. admin -epub'ı elle düzenleyip tekrar convert tetiklerse) zaten aynı sourceKey -ile tekrar dener. +ÇÖKERTMEZ, sadece uyarı basar. Kitabın kendisi (epub + main'e PR) bu +adımdan tamamen bağımsız. """ import os @@ -26,19 +27,19 @@ INGEST_TIMEOUT_SECONDS = 60 -def push_book(book_slug: str, epub_path: str, title: str, +def push_book(book_slug: str, chapters: list, title: str, author: str | None = None, description: str | None = None) -> None: """ - eptran-web'in /api/ingest'ine tek bir epub kitabı gönderir. + eptran-web'in /api/ingest'ine bir kitabın TÜM bölümlerini format=txt + olarak gönderir. + + chapters: convert.py'nin load_txt_chapters()'ından dönen liste — + her eleman {"title": ..., "body": ...} içerir. WEB_INGEST_URL : örn. https://eptran-web-virid.vercel.app/api/ingest WEB_INGEST_SECRET: eptran-web'deki INGEST_SECRET ile AYNI değer - (Vercel projesindeki env var'ın GitHub Actions - secret'ı olarak kopyası) - İkisinden biri eksikse (henüz kurulmamışsa) sessizce (ama görünür bir - uyarıyla) atlanır — kitabın epub'a ciltlenip main'e PR açılması buna - bağlı değil. + İkisinden biri eksikse sessizce (ama görünür bir uyarıyla) atlanır. """ url = os.environ.get("WEB_INGEST_URL") secret = os.environ.get("WEB_INGEST_SECRET") @@ -48,24 +49,31 @@ def push_book(book_slug: str, epub_path: str, title: str, "eptran-web'e gönderim atlandı (bkz. README kurulum adımı).") return - if not os.path.exists(epub_path): - print(f" Uyarı: {epub_path} bulunamadı — eptran-web'e gönderim atlandı.") + if not chapters: + print(" Uyarı: gönderilecek bölüm yok — eptran-web'e gönderim atlandı.") return - data = {"sourceKey": book_slug, "title": title, "format": "epub"} + data = {"sourceKey": book_slug, "title": title, "format": "txt"} if author: data["author"] = author if description: data["description"] = description + files = [] + for i, ch in enumerate(chapters): + # parseTxtChapter ilk satırı başlık olarak alıyor — orijinal + # dosyadaki "# " markdown işaretçisini burada BİLEREK atıyoruz, + # yoksa sitede başlıklar "# Editor's Notes" gibi çirkin görünür. + content = f"{ch['title']}\n\n{ch['body']}" + fname = f"{i + 1:03d}_{book_slug}.txt" + files.append(("files[]", (fname, content.encode("utf-8"), "text/plain"))) + try: - with open(epub_path, "rb") as f: - files = {"file": (os.path.basename(epub_path), f, "application/epub+zip")} - resp = requests.post( - url, data=data, files=files, - headers={"Authorization": f"Bearer {secret}"}, - timeout=INGEST_TIMEOUT_SECONDS, - ) + resp = requests.post( + url, data=data, files=files, + headers={"Authorization": f"Bearer {secret}"}, + timeout=INGEST_TIMEOUT_SECONDS, + ) except requests.RequestException as e: print(f" Uyarı: eptran-web'e gönderim başarısız (ağ hatası): {e}") return