From 6dc52dc270c98500c9c280167b2e6468f979e13c Mon Sep 17 00:00:00 2001 From: maruson08 Date: Tue, 29 Sep 2026 15:36:49 +0900 Subject: [PATCH 1/2] =?UTF-8?q?=E2=9C=A8[Feat]=20Add=20PDF=20to=20Text=20t?= =?UTF-8?q?ool?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- js/i18n.d.ts | 2 + js/i18n.js | 8 +- js/locales/pdf-to-text.js | 65 ++++++++++++++ scripts/site-routes.mjs | 3 +- scripts/typescript-modules.mjs | 15 ++++ sitemap.xml | 1 + tests/category-availability.test.mjs | 4 +- tests/i18n-quality.test.mjs | 4 +- tests/ocr-foundation.test.mjs | 2 +- tests/pdf-to-text.test.mjs | 98 +++++++++++++++++++++ tests/run-all.mjs | 1 + tests/seo-foundation.test.mjs | 2 +- tests/typescript-foundation.test.mjs | 2 +- tests/url-namespace.test.mjs | 2 +- tools/pdf/index.html | 1 + tools/pdf/split/pdf.d.ts | 1 + tools/pdf/to-text/app.ts | 125 +++++++++++++++++++++++++++ tools/pdf/to-text/controller.ts | 89 +++++++++++++++++++ tools/pdf/to-text/index.html | 16 ++++ tools/pdf/to-text/output.ts | 23 +++++ tools/pdf/to-text/tool.css | 15 ++++ tools/scan/index.html | 2 +- tools/shared/file.d.ts | 2 + tools/shared/pdf.d.ts | 1 + tsconfig.build.json | 3 + tsconfig.json | 5 +- 26 files changed, 478 insertions(+), 14 deletions(-) create mode 100644 js/i18n.d.ts create mode 100644 js/locales/pdf-to-text.js create mode 100644 tests/pdf-to-text.test.mjs create mode 100644 tools/pdf/split/pdf.d.ts create mode 100644 tools/pdf/to-text/app.ts create mode 100644 tools/pdf/to-text/controller.ts create mode 100644 tools/pdf/to-text/index.html create mode 100644 tools/pdf/to-text/output.ts create mode 100644 tools/pdf/to-text/tool.css create mode 100644 tools/shared/file.d.ts diff --git a/js/i18n.d.ts b/js/i18n.d.ts new file mode 100644 index 0000000..4eccda4 --- /dev/null +++ b/js/i18n.d.ts @@ -0,0 +1,2 @@ +export function t(key: string): string; +export function initializeI18n(): void; diff --git a/js/i18n.js b/js/i18n.js index 09f9365..9542849 100644 --- a/js/i18n.js +++ b/js/i18n.js @@ -8,6 +8,7 @@ import { imageResizeLocales } from "./locales/image-resize.js"; import { imageCompressorLocales } from "./locales/image-compressor.js"; import { imageMetadataLocales } from "./locales/image-metadata.js"; import { imageToTextLocales } from "./locales/image-to-text.js"; +import { pdfToTextLocales } from "./locales/pdf-to-text.js"; import { privacyHubLocales } from "./locales/privacy-hub.js"; import { metadataUxLocales } from "./locales/metadata-ux.js"; @@ -15,13 +16,14 @@ const STORAGE_KEY = "secure-tools-language"; const baseTranslations = { en, ko, ja, es, de, fr }; export const translations = Object.fromEntries(Object.entries(baseTranslations).map(([language, catalog]) => [language, { ...catalog, - metadata: { ...catalog.metadata, imageResize: imageResizeLocales[language].metadata, imageCompressor: imageCompressorLocales[language].metadata, imageMetadata: imageMetadataLocales[language].metadata, imageToText: imageToTextLocales[language].metadata, privacyCategory: privacyHubLocales[language].metadata }, - tools: { ...catalog.tools, imageMetadata: imageMetadataLocales[language].toolName, imageToText: imageToTextLocales[language].toolName, categoryDescriptions: { ...catalog.tools.categoryDescriptions, privacy: privacyHubLocales[language].categoryDescription } }, - categories: { ...catalog.categories, image: { ...catalog.categories.image, metadata: imageMetadataLocales[language].categoryDescription, toText: imageToTextLocales[language].categoryDescription } }, + metadata: { ...catalog.metadata, imageResize: imageResizeLocales[language].metadata, imageCompressor: imageCompressorLocales[language].metadata, imageMetadata: imageMetadataLocales[language].metadata, imageToText: imageToTextLocales[language].metadata, pdfToText: pdfToTextLocales[language].metadata, privacyCategory: privacyHubLocales[language].metadata }, + tools: { ...catalog.tools, imageMetadata: imageMetadataLocales[language].toolName, imageToText: imageToTextLocales[language].toolName, pdfToText: pdfToTextLocales[language].toolName, categoryDescriptions: { ...catalog.tools.categoryDescriptions, privacy: privacyHubLocales[language].categoryDescription } }, + categories: { ...catalog.categories, pdf: { ...catalog.categories.pdf, toText: pdfToTextLocales[language].categoryDescription }, image: { ...catalog.categories.image, metadata: imageMetadataLocales[language].categoryDescription, toText: imageToTextLocales[language].categoryDescription }, scan: { ...catalog.categories.scan, pdfToText: pdfToTextLocales[language].categoryDescription } }, imageResize: imageResizeLocales[language].copy, imageCompressor: imageCompressorLocales[language].copy, imageMetadata: { ...imageMetadataLocales[language].copy, source: { ...imageMetadataLocales[language].copy.source, ...metadataUxLocales[language].image.source }, inspector: { ...imageMetadataLocales[language].copy.inspector, ...metadataUxLocales[language].image.inspector }, clean: { ...imageMetadataLocales[language].copy.clean, ...metadataUxLocales[language].image.clean }, policy: metadataUxLocales[language].image.policy }, imageToText: imageToTextLocales[language].copy, + pdfToText: pdfToTextLocales[language].copy, pdfMetadata: { ...catalog.pdfMetadata, source: { ...catalog.pdfMetadata.source, ...metadataUxLocales[language].pdf.source }, inspector: { ...catalog.pdfMetadata.inspector, ...metadataUxLocales[language].pdf.inspector }, actions: { ...catalog.pdfMetadata.actions, ...metadataUxLocales[language].pdf.actions }, custom: metadataUxLocales[language].pdf.custom, errors: { ...catalog.pdfMetadata.errors, ...metadataUxLocales[language].pdf.errors } }, privacyHub: privacyHubLocales[language].copy, }])); diff --git a/js/locales/pdf-to-text.js b/js/locales/pdf-to-text.js new file mode 100644 index 0000000..2e7caa7 --- /dev/null +++ b/js/locales/pdf-to-text.js @@ -0,0 +1,65 @@ +const copy = { + eyebrow: "Local PDF OCR", title: "PDF to Text", description: "Recognize editable text from every page or a selected page range without uploading your PDF.", + drop: { title: "Add one PDF", description: "Drop a PDF here or use the picker.", choose: "Choose PDF", localTitle: "Processed locally.", localBody: "The PDF, rendered pages, and recognized text stay on this device.", privacyLink: "How privacy works" }, + source: { title: "Source PDF", empty: "No PDF selected yet.", selectedLabel: "Selected source PDF", meta: "{pages} pages · {size}", replace: "Replace PDF", remove: "Remove PDF" }, + warning: { large: "Large PDFs can take a while. Keep this tab open while local OCR runs." }, + settings: { title: "Recognition", pages: "Pages", all: "All pages", selected: "Selected pages", range: "Page range", rangePlaceholder: "1, 3-5", rangeHelp: "Use commas and ranges, such as 1, 3-5.", language: "Text language", english: "English", korean: "Korean", combined: "English + Korean", recognize: "Recognize text", cancel: "Cancel recognition", progress: "Recognition progress" }, + result: { title: "Recognized text", description: "Review and edit each page before copying or downloading it.", page: "Page {page}", pageLabel: "Editable recognized text for page {page}", copyPage: "Copy page", copyAll: "Copy all", download: "Download TXT" }, + status: { preparing: "Preparing the PDF locally…", ready: "PDF ready. Choose pages and a language to begin.", cancelled: "Recognition cancelled. The PDF remains ready.", success: "Recognition complete. You can edit every page.", copied: "All recognized text copied.", pageCopied: "Page {page} copied.", downloaded: "Downloaded {name}." }, + progress: { starting: "Starting local recognition…", loading: "Loading the PDF locally…", "rendering-page": "Rendering page {page} of {total}…", "recognizing-page": "Recognizing page {page} of {total}… {percent}%", complete: "Finishing recognition…" }, + errors: { oneFile: "Choose exactly one PDF.", UNSUPPORTED_PDF: "Choose a valid PDF file.", UNREADABLE_PDF: "This PDF could not be read.", ENCRYPTED_PDF: "Password-protected PDFs are not supported.", PDF_HAS_NO_PAGES: "This PDF has no pages.", PAGE_RANGE_REQUIRED: "Enter at least one page.", PAGE_RANGE_INVALID: "Enter pages as numbers and ranges, such as 1, 3-5.", PAGE_RANGE_REVERSED: "A page range cannot run backwards.", PAGE_OUT_OF_RANGE: "One or more pages are outside this PDF.", PDF_OCR_PAGE_DUPLICATE: "Choose each page only once.", PDF_OCR_RENDER_DIMENSION_EXCEEDED: "A page is too large to render safely.", PDF_OCR_RENDER_PIXELS_EXCEEDED: "A page is too large to render safely.", PDF_OCR_FAILED: "Recognition failed. Try again or choose another PDF.", PDF_LIBRARY_UNAVAILABLE: "The local PDF reader is unavailable.", copy: "The text could not be copied. Select it manually and copy it.", download: "The text file could not be downloaded." }, +}; +const localized = { + en: { title: "PDF to Text", description: copy.description, toolName: "PDF to Text OCR", category: "Extract editable text from PDF pages with local English and Korean OCR." }, + ko: { title: "PDF → 텍스트", description: "PDF를 업로드하지 않고 전체 또는 선택한 페이지에서 편집 가능한 텍스트를 인식합니다.", toolName: "PDF 텍스트 OCR", category: "로컬 영어·한국어 OCR로 PDF 페이지에서 편집 가능한 텍스트를 추출합니다." }, + ja: { title: "PDFからテキスト", description: "PDFをアップロードせず、全ページまたは選択ページから編集可能なテキストを認識します。", toolName: "PDFテキストOCR", category: "ローカルの英語・韓国語OCRでPDFページから編集可能なテキストを抽出します。" }, + es: { title: "PDF a texto", description: "Reconoce texto editable de todas las páginas o de una selección sin subir el PDF.", toolName: "OCR de PDF a texto", category: "Extrae texto editable de páginas PDF con OCR local en inglés y coreano." }, + de: { title: "PDF zu Text", description: "Erkennt bearbeitbaren Text auf allen oder ausgewählten Seiten, ohne die PDF hochzuladen.", toolName: "PDF-zu-Text-OCR", category: "Extrahiert bearbeitbaren Text aus PDF-Seiten mit lokaler englischer und koreanischer OCR." }, + fr: { title: "PDF en texte", description: "Reconnaît du texte modifiable dans toutes les pages ou une sélection sans téléverser le PDF.", toolName: "OCR PDF en texte", category: "Extrait du texte modifiable des pages PDF par OCR local anglais et coréen." }, +}; +const overrides = { + ko: { + eyebrow: "로컬 PDF OCR", drop: { title: "PDF 한 개 추가", description: "PDF를 놓거나 파일 선택기를 사용하세요.", choose: "PDF 선택", localTitle: "로컬에서 처리됩니다.", localBody: "PDF, 렌더링한 페이지, 인식된 텍스트는 이 기기에만 남습니다.", privacyLink: "개인정보 보호 방식" }, + source: { title: "원본 PDF", empty: "선택한 PDF가 없습니다.", selectedLabel: "선택한 원본 PDF", meta: "{pages}페이지 · {size}", replace: "PDF 교체", remove: "PDF 제거" }, warning: { large: "큰 PDF는 시간이 걸릴 수 있습니다. 로컬 OCR이 실행되는 동안 이 탭을 열어 두세요." }, + settings: { title: "텍스트 인식", pages: "페이지", all: "모든 페이지", selected: "선택한 페이지", range: "페이지 범위", rangePlaceholder: "1, 3-5", rangeHelp: "1, 3-5처럼 쉼표와 범위를 사용하세요.", language: "텍스트 언어", english: "영어", korean: "한국어", combined: "영어 + 한국어", recognize: "텍스트 인식", cancel: "인식 취소", progress: "인식 진행률" }, + result: { title: "인식된 텍스트", description: "복사하거나 다운로드하기 전에 각 페이지를 확인하고 편집하세요.", page: "{page}페이지", pageLabel: "{page}페이지의 편집 가능한 인식 텍스트", copyPage: "페이지 복사", copyAll: "모두 복사", download: "TXT 다운로드" }, + status: { preparing: "PDF를 로컬에서 준비하는 중…", ready: "PDF가 준비되었습니다. 페이지와 언어를 선택하세요.", cancelled: "인식을 취소했습니다. PDF는 준비 상태로 유지됩니다.", success: "인식이 완료되었습니다. 각 페이지를 편집할 수 있습니다.", copied: "인식된 텍스트를 모두 복사했습니다.", pageCopied: "{page}페이지를 복사했습니다.", downloaded: "{name}을 다운로드했습니다." }, + progress: { starting: "로컬 인식을 시작하는 중…", loading: "PDF를 로컬에서 불러오는 중…", "rendering-page": "{total}페이지 중 {page}페이지를 렌더링하는 중…", "recognizing-page": "{total}페이지 중 {page}페이지를 인식하는 중… {percent}%", complete: "인식을 마무리하는 중…" }, + }, + ja: { + eyebrow: "ローカルPDF OCR", drop: { title: "PDFを1件追加", description: "PDFをドロップするか、ファイル選択を使用します。", choose: "PDFを選択", localTitle: "ローカル処理。", localBody: "PDF、描画ページ、認識テキストは端末外へ送信されません。", privacyLink: "プライバシーの仕組み" }, + source: { title: "元のPDF", empty: "PDFが未選択です。", selectedLabel: "選択した元のPDF", meta: "{pages}ページ · {size}", replace: "PDFを置換", remove: "PDFを削除" }, warning: { large: "大きなPDFは時間がかかることがあります。ローカルOCR中はこのタブを開いたままにしてください。" }, + settings: { title: "文字認識", pages: "ページ", all: "全ページ", selected: "選択したページ", range: "ページ範囲", rangePlaceholder: "1, 3-5", rangeHelp: "1, 3-5のようにカンマと範囲を使用します。", language: "テキスト言語", english: "英語", korean: "韓国語", combined: "英語 + 韓国語", recognize: "テキストを認識", cancel: "認識をキャンセル", progress: "認識の進行状況" }, + result: { title: "認識したテキスト", description: "コピーまたはダウンロード前に各ページを確認・編集できます。", page: "ページ{page}", pageLabel: "ページ{page}の編集可能な認識テキスト", copyPage: "ページをコピー", copyAll: "すべてコピー", download: "TXTをダウンロード" }, + status: { preparing: "PDFをローカルで準備中…", ready: "PDFの準備ができました。ページと言語を選択してください。", cancelled: "認識をキャンセルしました。PDFは準備済みです。", success: "認識が完了しました。各ページを編集できます。", copied: "認識テキストをすべてコピーしました。", pageCopied: "ページ{page}をコピーしました。", downloaded: "{name}をダウンロードしました。" }, + progress: { starting: "ローカル認識を開始中…", loading: "PDFをローカルで読み込み中…", "rendering-page": "{total}ページ中{page}ページを描画中…", "recognizing-page": "{total}ページ中{page}ページを認識中… {percent}%", complete: "認識を完了中…" }, + }, + es: { + eyebrow: "OCR local de PDF", drop: { title: "Añadir un PDF", description: "Suelta un PDF aquí o usa el selector.", choose: "Elegir PDF", localTitle: "Procesamiento local.", localBody: "El PDF, las páginas renderizadas y el texto reconocido permanecen en este dispositivo.", privacyLink: "Cómo funciona la privacidad" }, + source: { title: "PDF de origen", empty: "No hay ningún PDF seleccionado.", selectedLabel: "PDF de origen seleccionado", meta: "{pages} páginas · {size}", replace: "Reemplazar PDF", remove: "Quitar PDF" }, warning: { large: "Los PDF grandes pueden tardar. Mantén esta pestaña abierta mientras se ejecuta el OCR local." }, + settings: { title: "Reconocimiento", pages: "Páginas", all: "Todas las páginas", selected: "Páginas seleccionadas", range: "Rango de páginas", rangePlaceholder: "1, 3-5", rangeHelp: "Usa comas y rangos, por ejemplo 1, 3-5.", language: "Idioma del texto", english: "Inglés", korean: "Coreano", combined: "Inglés + coreano", recognize: "Reconocer texto", cancel: "Cancelar reconocimiento", progress: "Progreso del reconocimiento" }, + result: { title: "Texto reconocido", description: "Revisa y edita cada página antes de copiarla o descargarla.", page: "Página {page}", pageLabel: "Texto reconocido editable de la página {page}", copyPage: "Copiar página", copyAll: "Copiar todo", download: "Descargar TXT" }, + status: { preparing: "Preparando el PDF localmente…", ready: "PDF listo. Elige páginas e idioma para comenzar.", cancelled: "Reconocimiento cancelado. El PDF sigue listo.", success: "Reconocimiento terminado. Puedes editar cada página.", copied: "Se copió todo el texto reconocido.", pageCopied: "Página {page} copiada.", downloaded: "Se descargó {name}." }, + progress: { starting: "Iniciando el reconocimiento local…", loading: "Cargando el PDF localmente…", "rendering-page": "Renderizando página {page} de {total}…", "recognizing-page": "Reconociendo página {page} de {total}… {percent}%", complete: "Finalizando el reconocimiento…" }, + }, + de: { + eyebrow: "Lokale PDF-OCR", drop: { title: "Eine PDF hinzufügen", description: "PDF hier ablegen oder Dateiauswahl verwenden.", choose: "PDF auswählen", localTitle: "Lokale Verarbeitung.", localBody: "PDF, gerenderte Seiten und erkannter Text bleiben auf diesem Gerät.", privacyLink: "So funktioniert der Datenschutz" }, + source: { title: "Quell-PDF", empty: "Keine PDF ausgewählt.", selectedLabel: "Ausgewählte Quell-PDF", meta: "{pages} Seiten · {size}", replace: "PDF ersetzen", remove: "PDF entfernen" }, warning: { large: "Große PDFs können länger dauern. Lassen Sie diesen Tab während der lokalen OCR geöffnet." }, + settings: { title: "Erkennung", pages: "Seiten", all: "Alle Seiten", selected: "Ausgewählte Seiten", range: "Seitenbereich", rangePlaceholder: "1, 3-5", rangeHelp: "Kommas und Bereiche verwenden, zum Beispiel 1, 3-5.", language: "Textsprache", english: "Englisch", korean: "Koreanisch", combined: "Englisch + Koreanisch", recognize: "Text erkennen", cancel: "Erkennung abbrechen", progress: "Erkennungsfortschritt" }, + result: { title: "Erkannter Text", description: "Jede Seite vor dem Kopieren oder Herunterladen prüfen und bearbeiten.", page: "Seite {page}", pageLabel: "Bearbeitbarer erkannter Text für Seite {page}", copyPage: "Seite kopieren", copyAll: "Alles kopieren", download: "TXT herunterladen" }, + status: { preparing: "PDF wird lokal vorbereitet…", ready: "PDF bereit. Seiten und Sprache auswählen.", cancelled: "Erkennung abgebrochen. Die PDF bleibt bereit.", success: "Erkennung abgeschlossen. Jede Seite kann bearbeitet werden.", copied: "Gesamten erkannten Text kopiert.", pageCopied: "Seite {page} kopiert.", downloaded: "{name} wurde heruntergeladen." }, + progress: { starting: "Lokale Erkennung wird gestartet…", loading: "PDF wird lokal geladen…", "rendering-page": "Seite {page} von {total} wird gerendert…", "recognizing-page": "Seite {page} von {total} wird erkannt… {percent}%", complete: "Erkennung wird abgeschlossen…" }, + }, + fr: { + eyebrow: "OCR PDF local", drop: { title: "Ajouter un PDF", description: "Déposez un PDF ici ou utilisez le sélecteur.", choose: "Choisir un PDF", localTitle: "Traitement local.", localBody: "Le PDF, les pages rendues et le texte reconnu restent sur cet appareil.", privacyLink: "Fonctionnement de la confidentialité" }, + source: { title: "PDF source", empty: "Aucun PDF sélectionné.", selectedLabel: "PDF source sélectionné", meta: "{pages} pages · {size}", replace: "Remplacer le PDF", remove: "Retirer le PDF" }, warning: { large: "Les PDF volumineux peuvent prendre du temps. Gardez cet onglet ouvert pendant l’OCR local." }, + settings: { title: "Reconnaissance", pages: "Pages", all: "Toutes les pages", selected: "Pages sélectionnées", range: "Plage de pages", rangePlaceholder: "1, 3-5", rangeHelp: "Utilisez des virgules et des plages, par exemple 1, 3-5.", language: "Langue du texte", english: "Anglais", korean: "Coréen", combined: "Anglais + coréen", recognize: "Reconnaître le texte", cancel: "Annuler la reconnaissance", progress: "Progression de la reconnaissance" }, + result: { title: "Texte reconnu", description: "Vérifiez et modifiez chaque page avant de la copier ou de la télécharger.", page: "Page {page}", pageLabel: "Texte reconnu modifiable de la page {page}", copyPage: "Copier la page", copyAll: "Tout copier", download: "Télécharger le TXT" }, + status: { preparing: "Préparation locale du PDF…", ready: "PDF prêt. Choisissez les pages et la langue.", cancelled: "Reconnaissance annulée. Le PDF reste prêt.", success: "Reconnaissance terminée. Chaque page est modifiable.", copied: "Tout le texte reconnu a été copié.", pageCopied: "Page {page} copiée.", downloaded: "{name} téléchargé." }, + progress: { starting: "Démarrage de la reconnaissance locale…", loading: "Chargement local du PDF…", "rendering-page": "Rendu de la page {page} sur {total}…", "recognizing-page": "Reconnaissance de la page {page} sur {total}… {percent}%", complete: "Finalisation de la reconnaissance…" }, + }, +}; +export const pdfToTextLocales = Object.fromEntries(Object.entries(localized).map(([language, value]) => [language, { + metadata: { title: `${value.toolName} — Secure Tools`, description: value.description }, toolName: value.toolName, + categoryDescription: value.category, copy: Object.fromEntries(Object.entries({ ...copy, title: value.title, description: value.description }).map(([key, item]) => [key, item && typeof item === "object" ? { ...item, ...(overrides[language]?.[key] || {}) } : (overrides[language]?.[key] || item)])), +}])); diff --git a/scripts/site-routes.mjs b/scripts/site-routes.mjs index 4465863..025ffe9 100644 --- a/scripts/site-routes.mjs +++ b/scripts/site-routes.mjs @@ -11,6 +11,7 @@ export const canonicalPages = [ { source: "tools/pdf/organize/index.html", route: "/pdf/organize/" }, { source: "tools/pdf/to-images/index.html", route: "/pdf/to-images/" }, { source: "tools/pdf/metadata/index.html", route: "/pdf/metadata/" }, + { source: "tools/pdf/to-text/index.html", route: "/pdf/to-text/", legacy: false }, { source: "tools/image/index.html", route: "/image/" }, { source: "tools/image/converter/index.html", route: "/image/converter/" }, { source: "tools/image/resize/index.html", route: "/image/resize/" }, @@ -23,7 +24,7 @@ export const canonicalPages = [ export const legacyRedirects = [ ...canonicalPages - .filter(({ source }) => source.startsWith("tools/")) + .filter(({ source, legacy }) => source.startsWith("tools/") && legacy !== false) .map(({ route }) => ({ from: `/tools${route}`, to: route })), { from: "/tools/privacy/", to: "/privacy/" }, { from: "/tools/image-to-pdf/", to: "/pdf/images-to-pdf/" }, diff --git a/scripts/typescript-modules.mjs b/scripts/typescript-modules.mjs index 6951c96..2205fbb 100644 --- a/scripts/typescript-modules.mjs +++ b/scripts/typescript-modules.mjs @@ -1,4 +1,19 @@ export const compiledBrowserModules = Object.freeze([ + Object.freeze({ + source: "tools/pdf/to-text/app.ts", + compiled: "tools/pdf/to-text/app.js", + public: "pdf/to-text/app.js", + }), + Object.freeze({ + source: "tools/pdf/to-text/controller.ts", + compiled: "tools/pdf/to-text/controller.js", + public: "pdf/to-text/controller.js", + }), + Object.freeze({ + source: "tools/pdf/to-text/output.ts", + compiled: "tools/pdf/to-text/output.js", + public: "pdf/to-text/output.js", + }), Object.freeze({ source: "tools/shared/pdf-ocr.ts", compiled: "tools/shared/pdf-ocr.js", diff --git a/sitemap.xml b/sitemap.xml index 574798e..e0fc784 100644 --- a/sitemap.xml +++ b/sitemap.xml @@ -10,6 +10,7 @@ https://tools.securetools.app/pdf/organize/ https://tools.securetools.app/pdf/to-images/ https://tools.securetools.app/pdf/metadata/ + https://tools.securetools.app/pdf/to-text/ https://tools.securetools.app/image/ https://tools.securetools.app/image/converter/ https://tools.securetools.app/image/resize/ diff --git a/tests/category-availability.test.mjs b/tests/category-availability.test.mjs index 2cdc228..141281f 100644 --- a/tests/category-availability.test.mjs +++ b/tests/category-availability.test.mjs @@ -84,8 +84,8 @@ assert.match(privacyHtml, /data-i18n="privacyHub\.pdfDescription"/); assertPublicRoutesExist("/privacy/", privacyRoutes); const pdfList = categoryList(read("tools/pdf/index.html")); -assert.equal(linkedRoutes(pdfList).length, 6, "Every PDF production card must remain linked"); -assert.equal((pdfList.match(/status--available/g) || []).length, 6); +assert.equal(linkedRoutes(pdfList).length, 7, "Every PDF production card must remain linked"); +assert.equal((pdfList.match(/status--available/g) || []).length, 7); for (const category of ["pdf", "image", "scan"]) { const html = read(`tools/${category}/index.html`); diff --git a/tests/i18n-quality.test.mjs b/tests/i18n-quality.test.mjs index 39f5a2f..baa84d0 100644 --- a/tests/i18n-quality.test.mjs +++ b/tests/i18n-quality.test.mjs @@ -46,7 +46,7 @@ function placeholders(value) { function testCatalogParityAndQuality() { assert.deepEqual([...Object.keys(translations)], [...languageNames.keys()]); const english = flatten(translations.en); - assert.equal(english.size, 817); + assert.equal(english.size, 887); for (const [language, catalog] of Object.entries(translations)) { const flattened = flatten(catalog); @@ -78,7 +78,7 @@ function testResolutionDetectionAndPersistence() { function testSelectorsAndDocumentTranslation() { const pages = canonicalPages.map(({ source }) => path.join(root, source)); - assert.equal(pages.length, 18, "Every canonical page comes from the route manifest"); + assert.equal(pages.length, 19, "Every canonical page comes from the route manifest"); for (const file of pages) { const html = fs.readFileSync(file, "utf8"); const select = html.match(/]*data-language-select[^>]*>([\s\S]*?)<\/select>/)?.[1]; diff --git a/tests/ocr-foundation.test.mjs b/tests/ocr-foundation.test.mjs index b11f2bd..9ef4817 100644 --- a/tests/ocr-foundation.test.mjs +++ b/tests/ocr-foundation.test.mjs @@ -210,7 +210,7 @@ const publicOcrReferences = listAbsoluteFiles(path.join(root, "tools")) .filter((file) => file.endsWith(".html")) .map((file) => path.relative(root, file).replaceAll("\\", "/")) .filter((relative) => read(relative).includes("assets/vendor/tesseract")); -assert.deepEqual(publicOcrReferences, ["tools/image/to-text/index.html"], "OCR runtime must stay lazy to its public route"); +assert.deepEqual(publicOcrReferences, ["tools/image/to-text/index.html", "tools/pdf/to-text/index.html"], "OCR runtime must stay lazy to public OCR routes"); for (const required of ["engine/tesseract.min.js", "worker/worker.min.js", "lang/eng.traineddata.gz", "lang/kor.traineddata.gz"]) { assert.ok(manifest.assets[required], `missing ${required}`); } diff --git a/tests/pdf-to-text.test.mjs b/tests/pdf-to-text.test.mjs new file mode 100644 index 0000000..762b6e6 --- /dev/null +++ b/tests/pdf-to-text.test.mjs @@ -0,0 +1,98 @@ +import assert from "node:assert/strict"; +import fs from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import { createPdfToTextController } from "../.ts-build/tools/pdf/to-text/controller.js"; +import { copyText, downloadPdfText, formatPdfOcrText, pdfTextFilename } from "../.ts-build/tools/pdf/to-text/output.js"; +import { canonicalPages, legacyRedirects } from "../scripts/site-routes.mjs"; +import { translations } from "../js/i18n.js"; + +const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), ".."); +const read = (file) => fs.readFileSync(path.join(root, file), "utf8"); +const html = read("tools/pdf/to-text/index.html"); + +assert.equal(canonicalPages.length, 19); +assert.equal(legacyRedirects.length, 17); +assert.ok(canonicalPages.some(({ route }) => route === "/pdf/to-text/")); +assert.ok(!legacyRedirects.some(({ from }) => from === "/tools/pdf/to-text/")); +assert.match(html, /data-page="pdfToText"/); +assert.match(html, /href="https:\/\/tools\.securetools\.app\/pdf\/to-text\/"/); +assert.match(html, /id="pages-all"[\s\S]*id="pages-selected"[\s\S]*id="page-range"/); +assert.match(html, /value="eng"[\s\S]*value="kor"[\s\S]*value="eng\+kor"/); +assert.match(html, /connect-src 'none'/); +assert.doesNotMatch(html, /https?:\/\/(?!tools\.securetools\.app|github\.com)/); +for (const language of ["en", "ko", "ja", "es", "de", "fr"]) { + assert.ok(translations[language].metadata.pdfToText.title); + assert.ok(translations[language].pdfToText.result.copyAll); + assert.ok(translations[language].categories.pdf.toText); +} + +assert.equal(formatPdfOcrText([{ pageNumber: 1, text: "One" }, { pageNumber: 3, text: "Three\n" }]), "--- Page 1 ---\nOne\n\n--- Page 3 ---\nThree\n"); +assert.equal(pdfTextFilename("private/report.pdf"), "private_report.txt"); + +let resolveFirst; +const first = new Promise((resolve) => { resolveFirst = resolve; }); +let request; +const states = []; +const service = { + async recognizeDocument(value) { request = value; await first; value.onPageResult?.({ pageNumber: 1, text: "stale" }); return { pages: [{ pageNumber: 1, text: "stale" }] }; }, + async cancel() { resolveFirst?.(); return true; }, + async dispose() {}, +}; +const controller = createPdfToTextController({ service, inspect: async () => ({ pageCount: 2 }), onChange: (state) => states.push(state.phase) }); +const file = Object.assign(new Blob(["%PDF"], { type: "application/pdf" }), { name: "sample.pdf", lastModified: 0 }); +await controller.select(file); +const running = controller.recognize("eng", { mode: "all" }); +await controller.cancel(); +await running; +assert.equal(controller.getState().phase, "cancelled"); +assert.deepEqual(controller.getState().pages, []); +assert.equal(request.selection.mode, "all"); +assert.ok(states.includes("recognizing")); + +let copied = ""; +await copyText(formatPdfOcrText([{ pageNumber: 2, text: "editable" }]), { navigatorObject: { clipboard: { writeText: async (text) => { copied = text; } } } }); +assert.equal(copied, "--- Page 2 ---\neditable\n"); +let downloaded; +const anchor = { click() { this.clicked = true; }, remove() {} }; +downloadPdfText([{ pageNumber: 2, text: "editable" }], "unsafe:name.pdf", { + documentObject: { body: { append() {} }, createElement: () => anchor }, + urlObject: { createObjectURL(blob) { downloaded = blob; return "blob:test"; }, revokeObjectURL() {} }, schedule() {}, +}); +assert.equal(anchor.download, "unsafe_name.txt"); +assert.equal(anchor.clicked, true); +assert.equal(await downloaded.text(), "--- Page 2 ---\neditable\n"); + +let attempts = 0; +const retryController = createPdfToTextController({ + inspect: async () => ({ pageCount: 1 }), + service: { + async recognizeDocument(value) { + attempts += 1; + if (attempts === 1) throw Object.assign(new Error("failed"), { code: "PDF_OCR_FAILED" }); + const page = { pageNumber: 1, text: "retry worked" }; value.onPageResult?.(page); + return { pages: [page] }; + }, + async cancel() { return false; }, async dispose() {}, + }, +}); +await retryController.select(file); +await retryController.recognize("eng", { mode: "all" }); +assert.equal(retryController.getState().phase, "error"); +await retryController.recognize("eng", { mode: "all" }); +assert.equal(retryController.getState().phase, "success"); +assert.equal(retryController.getState().pages[0].text, "retry worked"); + +let releaseOld; +const replacementController = createPdfToTextController({ + service: { async recognizeDocument() { throw new Error("unused"); }, async cancel() { return false; }, async dispose() {} }, + inspect: async (source) => source.name === "old.pdf" ? new Promise((resolve) => { releaseOld = () => resolve({ pageCount: 9 }); }) : { pageCount: 2 }, +}); +const oldFile = Object.assign(new Blob(["%PDF"], { type: "application/pdf" }), { name: "old.pdf", lastModified: 0 }); +const newFile = Object.assign(new Blob(["%PDF"], { type: "application/pdf" }), { name: "new.pdf", lastModified: 0 }); +const oldSelection = replacementController.select(oldFile); +await replacementController.select(newFile); +releaseOld(); await oldSelection; +assert.equal(replacementController.getState().source.file.name, "new.pdf"); +assert.equal(replacementController.getState().source.pageCount, 2); +console.log("PDF to Text route, localization, output, and stale-job cancellation checks passed."); diff --git a/tests/run-all.mjs b/tests/run-all.mjs index 5d4b8cb..09cf70f 100644 --- a/tests/run-all.mjs +++ b/tests/run-all.mjs @@ -32,6 +32,7 @@ for (const test of [ "tests/image-metadata.test.mjs", "tests/ocr-foundation.test.mjs", "tests/pdf-ocr-foundation.test.mjs", + "tests/pdf-to-text.test.mjs", "tests/image-to-text.test.mjs", "tests/typescript-foundation.test.mjs", "tests/category-availability.test.mjs", diff --git a/tests/seo-foundation.test.mjs b/tests/seo-foundation.test.mjs index 6501b86..beb4c61 100644 --- a/tests/seo-foundation.test.mjs +++ b/tests/seo-foundation.test.mjs @@ -15,7 +15,7 @@ const indexableRoutes = new Map(canonicalPages.map(({ source, route }) => [sourc const excludedRoutes = ["404.html", "tools/image-to-pdf/index.html"]; const allHtmlRoutes = [...indexableRoutes.keys(), ...excludedRoutes]; -assert.equal(indexableRoutes.size, 18, "all canonical pages come from the route manifest"); +assert.equal(indexableRoutes.size, 19, "all canonical pages come from the route manifest"); const shareImagePath = "assets/images/og-image.png"; const shareImageUrl = `${origin}/${shareImagePath}`; const iconLinks = new Map([ diff --git a/tests/typescript-foundation.test.mjs b/tests/typescript-foundation.test.mjs index 7180ac7..92ec1ba 100644 --- a/tests/typescript-foundation.test.mjs +++ b/tests/typescript-foundation.test.mjs @@ -21,7 +21,7 @@ assert.equal(typeConfig.compilerOptions.noEmit, true); assert.equal(typeConfig.compilerOptions.sourceMap, false); assert.equal(buildConfig.compilerOptions.outDir, ".ts-build"); -assert.equal(compiledBrowserModules.length, 4, "the compiled TypeScript boundary remains intentionally bounded"); +assert.equal(compiledBrowserModules.length, 7, "the compiled TypeScript boundary remains intentionally bounded"); assert.equal(new Set(compiledBrowserModules.map((module) => module.public)).size, compiledBrowserModules.length); for (const module of compiledBrowserModules) { assert.ok(fs.existsSync(path.join(root, module.source)), `${module.source} exists`); diff --git a/tests/url-namespace.test.mjs b/tests/url-namespace.test.mjs index c2865bd..47cc2c9 100644 --- a/tests/url-namespace.test.mjs +++ b/tests/url-namespace.test.mjs @@ -8,7 +8,7 @@ import { canonicalPages, legacyRedirects, productionOrigin, redirectStatus } fro const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), ".."); const routes = new Set(canonicalPages.map(({ route }) => route)); -assert.equal(canonicalPages.length, 18); +assert.equal(canonicalPages.length, 19); assert.equal(legacyRedirects.length, 17); assert.equal(redirectStatus, 308); assert.ok(routes.has("/image/to-text/")); diff --git a/tools/pdf/index.html b/tools/pdf/index.html index 2d527ae..64bdaf7 100644 --- a/tools/pdf/index.html +++ b/tools/pdf/index.html @@ -31,5 +31,6 @@
  • Available

    Organize PDF

    Preview, reorder, rotate, and remove pages from one PDF.

  • Available

    PDF to Images

    Convert ordered PDF pages to PNG, JPEG, or WebP images.

  • Available

    PDF Metadata

    Inspect common document metadata and create a verified cleaned copy.

  • +
  • Available

    PDF to Text OCR

    Extract editable text from PDF pages with local English and Korean OCR.

  • Available tools process file contents locally in browser memory.

    diff --git a/tools/pdf/split/pdf.d.ts b/tools/pdf/split/pdf.d.ts new file mode 100644 index 0000000..92c98af --- /dev/null +++ b/tools/pdf/split/pdf.d.ts @@ -0,0 +1 @@ +export function parsePageSelection(value: unknown, pageCount: number): number[]; diff --git a/tools/pdf/to-text/app.ts b/tools/pdf/to-text/app.ts new file mode 100644 index 0000000..b963259 --- /dev/null +++ b/tools/pdf/to-text/app.ts @@ -0,0 +1,125 @@ +import { t } from "../../../js/i18n.js"; +import { formatBytes } from "../../shared/file.js"; +import { inspectPdf } from "../../shared/pdf.js"; +import { PDF_OCR_LARGE_DOCUMENT_THRESHOLD } from "../../shared/pdf-ocr.js"; +import { parsePageSelection } from "../split/pdf.js"; +import { createPdfToTextController } from "./controller.js"; +import type { PdfToTextState } from "./controller.js"; +import { copyText, downloadPdfText, formatPdfOcrText, pdfTextFilename } from "./output.js"; + +declare global { interface Window { PDFLib?: { PDFDocument?: unknown } } } +type ElementMap = Record; +const byId = (id: string): T => { + const value = document.getElementById(id); + if (!value) throw new Error(`Missing #${id}`); + return value as T; +}; +const elements: ElementMap & { + input: HTMLInputElement; language: HTMLSelectElement; range: HTMLInputElement; + progress: HTMLProgressElement; all: HTMLInputElement; selected: HTMLInputElement; +} = { + input: byId("file-input"), language: byId("ocr-language"), + range: byId("page-range"), progress: byId("ocr-progress"), + all: byId("pages-all"), selected: byId("pages-selected"), + drop: byId("drop-zone"), sourceEmpty: byId("source-empty"), sourceCard: byId("source-card"), + sourceName: byId("source-name"), sourceMeta: byId("source-meta"), remove: byId("remove-source"), + replace: byId("replace-source"), warning: byId("large-warning"), rangePanel: byId("range-panel"), + rangeError: byId("range-error"), recognize: byId("recognize"), cancel: byId("cancel"), + status: byId("tool-status"), result: byId("result-panel"), pages: byId("result-pages"), + copyAll: byId("copy-all"), download: byId("download-result"), +}; + +const template = (key: string, values: Record = {}) => Object.entries(values).reduce( + (text, [name, value]) => text.replaceAll(`{${name}}`, String(value)), t(key), +); +const codeOf = (error: unknown) => error instanceof Error && "code" in error ? String(error.code) : "PDF_OCR_FAILED"; +let latest: PdfToTextState; + +function statusKey(state: PdfToTextState): string { + if (state.phase === "preparing") return "pdfToText.status.preparing"; + if (state.phase === "ready") return "pdfToText.status.ready"; + if (state.phase === "cancelled") return "pdfToText.status.cancelled"; + if (state.phase === "success") return "pdfToText.status.success"; + if (state.phase === "error") return `pdfToText.errors.${codeOf(state.error)}`; + return ""; +} + +function renderPages(state: PdfToTextState): void { + elements.pages.replaceChildren(); + state.pages.forEach((page) => { + const article = document.createElement("article"); article.className = "page-result surface"; + const header = document.createElement("div"); header.className = "page-result__header"; + const heading = document.createElement("h3"); heading.textContent = template("pdfToText.result.page", { page: page.pageNumber }); + const button = document.createElement("button"); button.className = "button button--secondary"; button.type = "button"; button.textContent = t("pdfToText.result.copyPage"); + const label = document.createElement("label"); const id = `page-text-${page.pageNumber}`; label.htmlFor = id; label.className = "visually-hidden"; label.textContent = template("pdfToText.result.pageLabel", { page: page.pageNumber }); + const field = document.createElement("textarea"); field.id = id; field.rows = 9; field.spellcheck = true; field.value = page.text; + field.addEventListener("input", () => controller.updatePageText(page.pageNumber, field.value)); + button.addEventListener("click", async () => { try { await copyText(field.value); showTransient("pdfToText.status.pageCopied", { page: page.pageNumber }); } catch { showError("pdfToText.errors.copy"); } }); + header.append(heading, button); article.append(header, label, field); elements.pages.append(article); + }); +} + +function progressText(state: PdfToTextState): string { + const progress = state.progress; + if (!progress) return t("pdfToText.progress.starting"); + if (progress.phase === "loading-document") return t("pdfToText.progress.loading"); + if (progress.phase === "complete") return t("pdfToText.progress.complete"); + return template(`pdfToText.progress.${progress.phase}`, { page: progress.pageNumber || 1, total: progress.pageCount || state.source?.pageCount || 1, percent: Math.round((progress.pageProgress || 0) * 100) }); +} + +function render(state: PdfToTextState): void { + latest = state; + const hasSource = Boolean(state.source); const busy = state.phase === "preparing" || state.phase === "recognizing"; + elements.sourceEmpty.hidden = hasSource; + elements.sourceCard.hidden = !hasSource; + elements.warning.hidden = !state.source || state.source.pageCount < PDF_OCR_LARGE_DOCUMENT_THRESHOLD; + if (state.source) { elements.sourceName.textContent = state.source.file.name; elements.sourceMeta.textContent = template("pdfToText.source.meta", { pages: state.source.pageCount, size: formatBytes(state.source.file.size) }); } + elements.input.toggleAttribute("disabled", busy); elements.language.toggleAttribute("disabled", busy); + elements.all.toggleAttribute("disabled", busy); elements.selected.toggleAttribute("disabled", busy); elements.range.toggleAttribute("disabled", busy || !elements.selected.checked); + elements.remove.toggleAttribute("disabled", busy); elements.replace.toggleAttribute("disabled", busy); + elements.recognize.toggleAttribute("disabled", !hasSource || busy); elements.recognize.hidden = state.phase === "recognizing"; + elements.cancel.hidden = state.phase !== "recognizing"; + elements.progress.hidden = state.phase !== "recognizing"; + if (state.phase === "recognizing") { + const value = state.progress?.overallProgress; + if (value === null || value === undefined) elements.progress.removeAttribute("value"); else elements.progress.value = value; + elements.status.textContent = progressText(state); + } else { elements.status.textContent = statusKey(state) ? t(statusKey(state)) : ""; } + elements.status.dataset.tone = state.phase === "error" ? "error" : state.phase === "success" ? "success" : state.phase === "cancelled" ? "warning" : ""; + elements.result.hidden = state.pages.length === 0; + renderPages(state); +} + +const controller = createPdfToTextController({ + inspect: (file) => inspectPdf(file, window.PDFLib?.PDFDocument), + onChange: render, +}); + +function selectedPages() { + if (elements.all.checked) return { mode: "all" as const }; + const pageNumbers = parsePageSelection(elements.range.value, latest.source?.pageCount || 0).map((page) => page + 1); + if (new Set(pageNumbers).size !== pageNumbers.length) throw Object.assign(new Error("PDF_OCR_PAGE_DUPLICATE"), { code: "PDF_OCR_PAGE_DUPLICATE" }); + return { mode: "selected" as const, pageNumbers }; +} +function clearRangeError(): void { elements.range.removeAttribute("aria-invalid"); elements.rangeError.textContent = ""; } +function showError(key: string): void { elements.status.textContent = t(key); elements.status.dataset.tone = "error"; } +function showTransient(key: string, values: Record = {}): void { elements.status.textContent = template(key, values); elements.status.dataset.tone = "success"; } +async function choose(files: FileList | File[]): Promise { + if (files.length !== 1) { showError("pdfToText.errors.oneFile"); return; } + clearRangeError(); await controller.select(files[0]); elements.input.value = ""; +} + +elements.input.addEventListener("change", () => { if (elements.input.files) void choose(elements.input.files); }); +elements.replace.addEventListener("click", () => elements.input.click()); elements.remove.addEventListener("click", () => void controller.reset()); +elements.cancel.addEventListener("click", () => void controller.cancel()); +elements.recognize.addEventListener("click", () => { clearRangeError(); try { void controller.recognize(elements.language.value as "eng" | "kor" | "eng+kor", selectedPages()); } catch (error) { elements.range.setAttribute("aria-invalid", "true"); elements.rangeError.textContent = t(`pdfToText.errors.${codeOf(error)}`); } }); +[elements.language, elements.all, elements.selected].forEach((element) => element.addEventListener("change", () => { elements.rangePanel.hidden = !elements.selected.checked; clearRangeError(); controller.invalidateResults(); })); +elements.range.addEventListener("input", () => { clearRangeError(); controller.invalidateResults(); }); +elements.copyAll.addEventListener("click", async () => { try { await copyText(formatPdfOcrText(latest.pages)); showTransient("pdfToText.status.copied"); } catch { showError("pdfToText.errors.copy"); } }); +elements.download.addEventListener("click", () => { try { downloadPdfText(latest.pages, latest.source?.file.name); showTransient("pdfToText.status.downloaded", { name: pdfTextFilename(latest.source?.file.name) }); } catch { showError("pdfToText.errors.download"); } }); +for (const type of ["dragenter", "dragover"]) elements.drop.addEventListener(type, (event) => { event.preventDefault(); elements.drop.dataset.dragging = "true"; }); +for (const type of ["dragleave", "drop"]) elements.drop.addEventListener(type, (event) => { event.preventDefault(); delete elements.drop.dataset.dragging; }); +elements.drop.addEventListener("drop", (event) => { const data = (event as DragEvent).dataTransfer; if (data) void choose(data.files); }); +document.addEventListener("securetools:languagechange", () => render(latest)); +window.addEventListener("beforeunload", () => void controller.dispose()); +render(controller.getState()); diff --git a/tools/pdf/to-text/controller.ts b/tools/pdf/to-text/controller.ts new file mode 100644 index 0000000..b64eebf --- /dev/null +++ b/tools/pdf/to-text/controller.ts @@ -0,0 +1,89 @@ +import { createPdfOcrService, readPdfOcrSource } from "../../shared/pdf-ocr.js"; +import type { PdfOcrPageResult, PdfOcrPageSelection, PdfOcrProgress, PdfOcrService } from "../../shared/pdf-ocr.js"; + +export type PdfToTextPhase = "empty" | "preparing" | "ready" | "recognizing" | "success" | "error" | "cancelled"; + +export interface PdfToTextSource { file: File; bytes: ArrayBuffer; pageCount: number } +export interface PdfToTextState { + phase: PdfToTextPhase; + source: PdfToTextSource | null; + pages: readonly PdfOcrPageResult[]; + progress: PdfOcrProgress | null; + error: unknown; +} +export interface PdfToTextControllerOptions { + service?: PdfOcrService; + inspect(file: File): Promise<{ pageCount: number }>; + onChange?(state: PdfToTextState): void; +} + +export function createPdfToTextController(options: PdfToTextControllerOptions) { + const service = options.service || createPdfOcrService(); + let generation = 0; + let state: PdfToTextState = { phase: "empty", source: null, pages: [], progress: null, error: null }; + const publish = (patch: Partial) => { + state = Object.freeze({ ...state, ...patch }); + options.onChange?.(state); + }; + const current = (token: number) => token === generation; + + async function select(file: File): Promise { + const token = ++generation; + await service.cancel(); + publish({ phase: "preparing", source: null, pages: [], progress: null, error: null }); + try { + const [bytes, details] = await Promise.all([readPdfOcrSource(file), options.inspect(file)]); + if (!current(token)) return; + publish({ phase: "ready", source: { file, bytes, pageCount: details.pageCount }, error: null }); + } catch (error) { + if (!current(token)) return; + publish({ phase: "error", source: null, error }); + } + } + + async function recognize(language: "eng" | "kor" | "eng+kor", selection: PdfOcrPageSelection): Promise { + if (!state.source || state.phase === "recognizing") return; + const token = ++generation; + const source = state.source; + publish({ phase: "recognizing", pages: [], progress: null, error: null }); + try { + const result = await service.recognizeDocument({ + sourceBytes: source.bytes, + language, + selection, + onProgress(progress) { if (current(token)) publish({ progress }); }, + onPageResult(page) { if (current(token)) publish({ pages: [...state.pages, page] }); }, + }); + if (!current(token)) return; + publish({ phase: "success", pages: result.pages, error: null }); + } catch (error) { + if (!current(token)) return; + const code = error instanceof Error && "code" in error ? String(error.code) : ""; + publish({ phase: code === "PDF_OCR_CANCELLED" ? "cancelled" : "error", progress: null, error }); + } + } + + async function cancel(): Promise { + if (state.phase !== "recognizing") return; + generation += 1; + await service.cancel(); + publish({ phase: "cancelled", progress: null, error: null }); + } + + async function reset(): Promise { + generation += 1; + await service.cancel(); + publish({ phase: "empty", source: null, pages: [], progress: null, error: null }); + } + + function invalidateResults(): void { + if (!state.source || state.phase === "recognizing") return; + publish({ phase: "ready", pages: [], progress: null, error: null }); + } + + function updatePageText(pageNumber: number, text: string): void { + publish({ pages: state.pages.map((page) => page.pageNumber === pageNumber ? { ...page, text } : page) }); + } + + return Object.freeze({ select, recognize, cancel, reset, invalidateResults, updatePageText, getState: () => state, dispose: () => service.dispose() }); +} diff --git a/tools/pdf/to-text/index.html b/tools/pdf/to-text/index.html new file mode 100644 index 0000000..2e4c1ce --- /dev/null +++ b/tools/pdf/to-text/index.html @@ -0,0 +1,16 @@ + + + + +PDF to Text OCR — Secure Tools + + + + +

    Local PDF OCR

    PDF to Text

    Recognize editable text from every page or a selected page range without uploading your PDF.

    +

    Add one PDF

    Drop a PDF here or use the picker.

    Processed locally. The PDF, rendered pages, and recognized text stay on this device. How privacy works

    +

    Source PDF

    No PDF selected yet.

    +
    +
    + + diff --git a/tools/pdf/to-text/output.ts b/tools/pdf/to-text/output.ts new file mode 100644 index 0000000..9244a74 --- /dev/null +++ b/tools/pdf/to-text/output.ts @@ -0,0 +1,23 @@ +import { downloadBlob } from "../../shared/save.js"; +import type { DownloadEnvironment } from "../../shared/save.js"; +import type { CopyEnvironment } from "../../image/to-text/output.js"; +import { copyText } from "../../image/to-text/output.js"; +import type { PdfOcrPageResult } from "../../shared/pdf-ocr.js"; + +const ILLEGAL_FILENAME_CHARACTERS = /[\\/:*?"<>|\u0000-\u001f]+/g; + +export function formatPdfOcrText(pages: readonly PdfOcrPageResult[]): string { + return pages.map(({ pageNumber, text }) => `--- Page ${pageNumber} ---\n${String(text).trimEnd()}`).join("\n\n") + (pages.length ? "\n" : ""); +} + +export function pdfTextFilename(sourceName: unknown): string { + const base = String(sourceName || "").replace(/\.pdf$/i, "").trim().replace(ILLEGAL_FILENAME_CHARACTERS, "_").replace(/[. ]+$/g, ""); + return `${base || "recognized-document"}.txt`; +} + +export { copyText }; +export function downloadPdfText(pages: readonly PdfOcrPageResult[], sourceName: unknown, environment?: DownloadEnvironment): void { + const blob = new Blob([formatPdfOcrText(pages)], { type: "text/plain;charset=utf-8" }); + downloadBlob(blob, pdfTextFilename(sourceName), environment); +} +export type { CopyEnvironment }; diff --git a/tools/pdf/to-text/tool.css b/tools/pdf/to-text/tool.css new file mode 100644 index 0000000..886495e --- /dev/null +++ b/tools/pdf/to-text/tool.css @@ -0,0 +1,15 @@ +.mode-list { grid-template-columns: 1fr; } +[hidden] { display: none !important; } +.mode-list legend { margin-bottom: var(--space-2); color: var(--text-secondary); font-size: .78rem; font-weight: var(--font-weight-bold); } +.source-actions, .result-actions { display: flex; flex-wrap: wrap; gap: var(--space-2); } +.tool-warning { margin: var(--space-4) 0 0; padding: var(--space-3) var(--space-4); border: 1px solid var(--tool-warning); border-radius: var(--radius-sm); background: var(--tool-warning-soft); color: var(--tool-warning); } +.result-summary { display: flex; align-items: center; justify-content: space-between; gap: var(--space-4); padding: var(--space-5); } +.result-summary h2 { margin-bottom: var(--space-2); } +.result-summary p { margin: 0; color: var(--text-secondary); } +.result-pages { display: grid; gap: var(--space-4); margin-top: var(--space-4); } +.page-result { padding: var(--space-5); } +.page-result__header { display: flex; align-items: center; justify-content: space-between; gap: var(--space-3); margin-bottom: var(--space-3); } +.page-result h3 { margin: 0; font-size: 1.1rem; } +.page-result textarea { width: 100%; min-height: 12rem; resize: vertical; padding: var(--space-3); border: 1px solid var(--border); border-radius: var(--radius-sm); background: var(--bg-primary); color: var(--text-primary); font: inherit; line-height: 1.55; } +.clipboard-fallback { position: fixed; left: -9999px; } +@media (max-width: 42rem) { .result-summary, .page-result__header { align-items: stretch; flex-direction: column; } .result-actions .button, .page-result__header .button { width: 100%; } } diff --git a/tools/scan/index.html b/tools/scan/index.html index 2830f61..eefaf9c 100644 --- a/tools/scan/index.html +++ b/tools/scan/index.html @@ -21,4 +21,4 @@ Scan and OCR tools — Secure Tools -

    Tool category

    Scan & OCR tools

    Turn scans into useful documents while keeping source material on your device.

    Available tools process file contents locally in browser memory.

    +

    Tool category

    Scan & OCR tools

    Turn scans into useful documents while keeping source material on your device.

    Available tools process file contents locally in browser memory.

    diff --git a/tools/shared/file.d.ts b/tools/shared/file.d.ts new file mode 100644 index 0000000..9a1807f --- /dev/null +++ b/tools/shared/file.d.ts @@ -0,0 +1,2 @@ +export function formatBytes(bytes: unknown): string; +export function sanitizePdfFilename(value: unknown, fallback?: string): string; diff --git a/tools/shared/pdf.d.ts b/tools/shared/pdf.d.ts index 18e3307..4ca016b 100644 --- a/tools/shared/pdf.d.ts +++ b/tools/shared/pdf.d.ts @@ -3,3 +3,4 @@ export interface PdfFileLike extends Blob { } export function isSupportedPdf(file: unknown): boolean; +export function inspectPdf(file: PdfFileLike, PDFDocument: unknown): Promise<{ pageCount: number }>; diff --git a/tsconfig.build.json b/tsconfig.build.json index 072290c..4521e33 100644 --- a/tsconfig.build.json +++ b/tsconfig.build.json @@ -9,6 +9,9 @@ "tools/image/to-text/controller.ts", "tools/image/to-text/output.ts", "tools/image/to-text/preview.ts", + "tools/pdf/to-text/app.ts", + "tools/pdf/to-text/controller.ts", + "tools/pdf/to-text/output.ts", "tools/shared/pdf-ocr.ts", "tools/shared/*.d.ts" ] diff --git a/tsconfig.json b/tsconfig.json index 1f8c68d..77c1eff 100644 --- a/tsconfig.json +++ b/tsconfig.json @@ -16,7 +16,10 @@ }, "include": [ "tools/image/to-text/*.ts", + "tools/pdf/to-text/*.ts", "tools/shared/pdf-ocr.ts", - "tools/shared/*.d.ts" + "tools/shared/*.d.ts", + "tools/pdf/split/*.d.ts", + "js/*.d.ts" ] } From de4ebfd98a62d47ae2bbc681797c96dc9271d165 Mon Sep 17 00:00:00 2001 From: maruson08 Date: Tue, 29 Sep 2026 15:44:20 +0900 Subject: [PATCH 2/2] =?UTF-8?q?=E2=9C=85[Test]=20Cover=20PDF=20to=20Text?= =?UTF-8?q?=20browser=20workflows?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- tests/browser/pdf-to-text-smoke.html | 3 ++ tests/browser/pdf-to-text-smoke.js | 59 ++++++++++++++++++++++++++++ tests/serve-ocr-smoke.mjs | 1 + 3 files changed, 63 insertions(+) create mode 100644 tests/browser/pdf-to-text-smoke.html create mode 100644 tests/browser/pdf-to-text-smoke.js diff --git a/tests/browser/pdf-to-text-smoke.html b/tests/browser/pdf-to-text-smoke.html new file mode 100644 index 0000000..06e6d9a --- /dev/null +++ b/tests/browser/pdf-to-text-smoke.html @@ -0,0 +1,3 @@ + +PDF to Text browser smoke +

    PDF to Text browser smoke

    Running…

      diff --git a/tests/browser/pdf-to-text-smoke.js b/tests/browser/pdf-to-text-smoke.js new file mode 100644 index 0000000..ee89b71 --- /dev/null +++ b/tests/browser/pdf-to-text-smoke.js @@ -0,0 +1,59 @@ +import { createPdfToTextController } from "/pdf/to-text/controller.js"; +import { copyText, downloadPdfText, formatPdfOcrText } from "/pdf/to-text/output.js"; +import { translations } from "/js/i18n.js"; + +const status = document.querySelector("#status"); +const results = document.querySelector("#results"); +const checks = []; +const assert = (condition, message) => { if (!condition) throw new Error(message); checks.push(message); }; +const file = (name) => new File(["%PDF-1.7"], name, { type: "application/pdf" }); +const show = () => { results.replaceChildren(...checks.map((message) => Object.assign(document.createElement("li"), { textContent: message }))); }; + +try { + const happy = createPdfToTextController({ + inspect: async () => ({ pageCount: 2 }), + service: { + async recognizeDocument(request) { + const pages = [{ pageNumber: 1, text: "first" }, { pageNumber: 2, text: "second" }]; + pages.forEach((page, index) => { request.onProgress?.({ phase: "recognizing-page", pageNumber: page.pageNumber, pageIndex: index + 1, pageCount: 2, documentPageCount: 2, pageProgress: 1, overallProgress: (index + 1) / 2, ocrStage: "recognizing" }); request.onPageResult?.(page); }); + return { pageCount: 2, selectedPageNumbers: [1, 2], pages, largeDocument: false }; + }, async cancel() { return false; }, async dispose() {}, + }, + }); + await happy.select(file("multi.pdf")); await happy.recognize("eng", { mode: "all" }); + assert(happy.getState().phase === "success", "happy path reaches success"); + assert(happy.getState().pages.map(({ pageNumber }) => pageNumber).join(",") === "1,2", "multi-page order is preserved"); + + let release; + const cancelled = createPdfToTextController({ inspect: async () => ({ pageCount: 1 }), service: { + async recognizeDocument() { await new Promise((resolve) => { release = resolve; }); throw Object.assign(new Error("cancelled"), { code: "PDF_OCR_CANCELLED" }); }, + async cancel() { release?.(); return true; }, async dispose() {}, + } }); + await cancelled.select(file("cancel.pdf")); const job = cancelled.recognize("eng", { mode: "all" }); await cancelled.cancel(); await job; + assert(cancelled.getState().phase === "cancelled" && cancelled.getState().pages.length === 0, "cancellation suppresses stale results"); + + let attempts = 0; + const retry = createPdfToTextController({ inspect: async () => ({ pageCount: 1 }), service: { + async recognizeDocument() { attempts += 1; if (attempts === 1) throw Object.assign(new Error("failed"), { code: "PDF_OCR_FAILED" }); return { pages: [{ pageNumber: 1, text: "recovered" }] }; }, + async cancel() { return false; }, async dispose() {}, + } }); + await retry.select(file("retry.pdf")); await retry.recognize("eng", { mode: "all" }); await retry.recognize("eng", { mode: "all" }); + assert(retry.getState().phase === "success", "retry recovers after a controlled failure"); + + let releaseOld; + const replacement = createPdfToTextController({ service: { async recognizeDocument() {}, async cancel() { return false; }, async dispose() {} }, inspect: async (source) => source.name === "old.pdf" ? new Promise((resolve) => { releaseOld = () => resolve({ pageCount: 9 }); }) : { pageCount: 2 } }); + const oldJob = replacement.select(file("old.pdf")); await replacement.select(file("new.pdf")); releaseOld(); await oldJob; + assert(replacement.getState().source.file.name === "new.pdf", "source replacement rejects stale preparation"); + + let clipboard = ""; + const text = formatPdfOcrText(happy.getState().pages); + await copyText(text, { navigatorObject: { clipboard: { writeText: async (value) => { clipboard = value; } } } }); + assert(clipboard.includes("--- Page 1 ---") && clipboard.includes("--- Page 2 ---"), "copy-all preserves page boundaries"); + let blob; const anchor = { click() { this.clicked = true; }, remove() {} }; + downloadPdfText(happy.getState().pages, "multi.pdf", { documentObject: { body: { append() {} }, createElement: () => anchor }, urlObject: { createObjectURL(value) { blob = value; return "blob:test"; }, revokeObjectURL() {} }, schedule() {} }); + assert(anchor.download === "multi.txt" && (await blob.text()) === text, "TXT export preserves filename and UTF-8 content"); + assert(["en", "ko", "ja", "es", "de", "fr"].every((language) => translations[language].pdfToText.title && translations[language].pdfToText.result.copyAll), "all six locale catalogs load without raw keys"); + show(); status.textContent = `${checks.length} browser checks passed.`; document.body.dataset.state = "passed"; +} catch (error) { + status.textContent = error instanceof Error ? error.message : String(error); document.body.dataset.state = "failed"; console.error(error); +} diff --git a/tests/serve-ocr-smoke.mjs b/tests/serve-ocr-smoke.mjs index b1d098a..ac86357 100644 --- a/tests/serve-ocr-smoke.mjs +++ b/tests/serve-ocr-smoke.mjs @@ -52,5 +52,6 @@ const server = http.createServer((request, response) => { server.listen(port, "127.0.0.1", () => { console.log(`OCR browser smoke: http://127.0.0.1:${port}/tests/browser/ocr-smoke.html`); console.log(`PDF OCR browser smoke: http://127.0.0.1:${port}/tests/browser/pdf-ocr-smoke.html`); + console.log(`PDF to Text browser smoke: http://127.0.0.1:${port}/tests/browser/pdf-to-text-smoke.html`); console.log(`Image to Text UI QA: http://127.0.0.1:${port}/image/to-text/`); });