Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions js/i18n.d.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,2 @@
export function t(key: string): string;
export function initializeI18n(): void;
8 changes: 5 additions & 3 deletions js/i18n.js
Original file line number Diff line number Diff line change
Expand Up @@ -8,20 +8,22 @@ import { imageResizeLocales } from "./locales/image-resize.js";
import { imageCompressorLocales } from "./locales/image-compressor.js";
import { imageMetadataLocales } from "./locales/image-metadata.js";
import { imageToTextLocales } from "./locales/image-to-text.js";
import { pdfToTextLocales } from "./locales/pdf-to-text.js";
import { privacyHubLocales } from "./locales/privacy-hub.js";
import { metadataUxLocales } from "./locales/metadata-ux.js";

const STORAGE_KEY = "secure-tools-language";
const baseTranslations = { en, ko, ja, es, de, fr };
export const translations = Object.fromEntries(Object.entries(baseTranslations).map(([language, catalog]) => [language, {
...catalog,
metadata: { ...catalog.metadata, imageResize: imageResizeLocales[language].metadata, imageCompressor: imageCompressorLocales[language].metadata, imageMetadata: imageMetadataLocales[language].metadata, imageToText: imageToTextLocales[language].metadata, privacyCategory: privacyHubLocales[language].metadata },
tools: { ...catalog.tools, imageMetadata: imageMetadataLocales[language].toolName, imageToText: imageToTextLocales[language].toolName, categoryDescriptions: { ...catalog.tools.categoryDescriptions, privacy: privacyHubLocales[language].categoryDescription } },
categories: { ...catalog.categories, image: { ...catalog.categories.image, metadata: imageMetadataLocales[language].categoryDescription, toText: imageToTextLocales[language].categoryDescription } },
metadata: { ...catalog.metadata, imageResize: imageResizeLocales[language].metadata, imageCompressor: imageCompressorLocales[language].metadata, imageMetadata: imageMetadataLocales[language].metadata, imageToText: imageToTextLocales[language].metadata, pdfToText: pdfToTextLocales[language].metadata, privacyCategory: privacyHubLocales[language].metadata },
tools: { ...catalog.tools, imageMetadata: imageMetadataLocales[language].toolName, imageToText: imageToTextLocales[language].toolName, pdfToText: pdfToTextLocales[language].toolName, categoryDescriptions: { ...catalog.tools.categoryDescriptions, privacy: privacyHubLocales[language].categoryDescription } },
categories: { ...catalog.categories, pdf: { ...catalog.categories.pdf, toText: pdfToTextLocales[language].categoryDescription }, image: { ...catalog.categories.image, metadata: imageMetadataLocales[language].categoryDescription, toText: imageToTextLocales[language].categoryDescription }, scan: { ...catalog.categories.scan, pdfToText: pdfToTextLocales[language].categoryDescription } },
imageResize: imageResizeLocales[language].copy,
imageCompressor: imageCompressorLocales[language].copy,
imageMetadata: { ...imageMetadataLocales[language].copy, source: { ...imageMetadataLocales[language].copy.source, ...metadataUxLocales[language].image.source }, inspector: { ...imageMetadataLocales[language].copy.inspector, ...metadataUxLocales[language].image.inspector }, clean: { ...imageMetadataLocales[language].copy.clean, ...metadataUxLocales[language].image.clean }, policy: metadataUxLocales[language].image.policy },
imageToText: imageToTextLocales[language].copy,
pdfToText: pdfToTextLocales[language].copy,
pdfMetadata: { ...catalog.pdfMetadata, source: { ...catalog.pdfMetadata.source, ...metadataUxLocales[language].pdf.source }, inspector: { ...catalog.pdfMetadata.inspector, ...metadataUxLocales[language].pdf.inspector }, actions: { ...catalog.pdfMetadata.actions, ...metadataUxLocales[language].pdf.actions }, custom: metadataUxLocales[language].pdf.custom, errors: { ...catalog.pdfMetadata.errors, ...metadataUxLocales[language].pdf.errors } },
privacyHub: privacyHubLocales[language].copy,
}]));
Expand Down
65 changes: 65 additions & 0 deletions js/locales/pdf-to-text.js

Large diffs are not rendered by default.

3 changes: 2 additions & 1 deletion scripts/site-routes.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@ export const canonicalPages = [
{ source: "tools/pdf/organize/index.html", route: "/pdf/organize/" },
{ source: "tools/pdf/to-images/index.html", route: "/pdf/to-images/" },
{ source: "tools/pdf/metadata/index.html", route: "/pdf/metadata/" },
{ source: "tools/pdf/to-text/index.html", route: "/pdf/to-text/", legacy: false },
{ source: "tools/image/index.html", route: "/image/" },
{ source: "tools/image/converter/index.html", route: "/image/converter/" },
{ source: "tools/image/resize/index.html", route: "/image/resize/" },
Expand All @@ -23,7 +24,7 @@ export const canonicalPages = [

export const legacyRedirects = [
...canonicalPages
.filter(({ source }) => source.startsWith("tools/"))
.filter(({ source, legacy }) => source.startsWith("tools/") && legacy !== false)
.map(({ route }) => ({ from: `/tools${route}`, to: route })),
{ from: "/tools/privacy/", to: "/privacy/" },
{ from: "/tools/image-to-pdf/", to: "/pdf/images-to-pdf/" },
Expand Down
15 changes: 15 additions & 0 deletions scripts/typescript-modules.mjs
Original file line number Diff line number Diff line change
@@ -1,4 +1,19 @@
export const compiledBrowserModules = Object.freeze([
Object.freeze({
source: "tools/pdf/to-text/app.ts",
compiled: "tools/pdf/to-text/app.js",
public: "pdf/to-text/app.js",
}),
Object.freeze({
source: "tools/pdf/to-text/controller.ts",
compiled: "tools/pdf/to-text/controller.js",
public: "pdf/to-text/controller.js",
}),
Object.freeze({
source: "tools/pdf/to-text/output.ts",
compiled: "tools/pdf/to-text/output.js",
public: "pdf/to-text/output.js",
}),
Object.freeze({
source: "tools/shared/pdf-ocr.ts",
compiled: "tools/shared/pdf-ocr.js",
Expand Down
1 change: 1 addition & 0 deletions sitemap.xml
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,7 @@
<url><loc>https://tools.securetools.app/pdf/organize/</loc></url>
<url><loc>https://tools.securetools.app/pdf/to-images/</loc></url>
<url><loc>https://tools.securetools.app/pdf/metadata/</loc></url>
<url><loc>https://tools.securetools.app/pdf/to-text/</loc></url>
<url><loc>https://tools.securetools.app/image/</loc></url>
<url><loc>https://tools.securetools.app/image/converter/</loc></url>
<url><loc>https://tools.securetools.app/image/resize/</loc></url>
Expand Down
3 changes: 3 additions & 0 deletions tests/browser/pdf-to-text-smoke.html
Original file line number Diff line number Diff line change
@@ -0,0 +1,3 @@
<!doctype html>
<html lang="en"><head><meta charset="utf-8"><meta name="viewport" content="width=device-width, initial-scale=1"><title>PDF to Text browser smoke</title></head>
<body><main><h1>PDF to Text browser smoke</h1><p id="status" role="status">Running…</p><ol id="results"></ol></main><script type="module" src="./pdf-to-text-smoke.js"></script></body></html>
59 changes: 59 additions & 0 deletions tests/browser/pdf-to-text-smoke.js
Original file line number Diff line number Diff line change
@@ -0,0 +1,59 @@
import { createPdfToTextController } from "/pdf/to-text/controller.js";
import { copyText, downloadPdfText, formatPdfOcrText } from "/pdf/to-text/output.js";
import { translations } from "/js/i18n.js";

const status = document.querySelector("#status");
const results = document.querySelector("#results");
const checks = [];
const assert = (condition, message) => { if (!condition) throw new Error(message); checks.push(message); };
const file = (name) => new File(["%PDF-1.7"], name, { type: "application/pdf" });
const show = () => { results.replaceChildren(...checks.map((message) => Object.assign(document.createElement("li"), { textContent: message }))); };

try {
const happy = createPdfToTextController({
inspect: async () => ({ pageCount: 2 }),
service: {
async recognizeDocument(request) {
const pages = [{ pageNumber: 1, text: "first" }, { pageNumber: 2, text: "second" }];
pages.forEach((page, index) => { request.onProgress?.({ phase: "recognizing-page", pageNumber: page.pageNumber, pageIndex: index + 1, pageCount: 2, documentPageCount: 2, pageProgress: 1, overallProgress: (index + 1) / 2, ocrStage: "recognizing" }); request.onPageResult?.(page); });
return { pageCount: 2, selectedPageNumbers: [1, 2], pages, largeDocument: false };
}, async cancel() { return false; }, async dispose() {},
},
});
await happy.select(file("multi.pdf")); await happy.recognize("eng", { mode: "all" });
assert(happy.getState().phase === "success", "happy path reaches success");
assert(happy.getState().pages.map(({ pageNumber }) => pageNumber).join(",") === "1,2", "multi-page order is preserved");

let release;
const cancelled = createPdfToTextController({ inspect: async () => ({ pageCount: 1 }), service: {
async recognizeDocument() { await new Promise((resolve) => { release = resolve; }); throw Object.assign(new Error("cancelled"), { code: "PDF_OCR_CANCELLED" }); },
async cancel() { release?.(); return true; }, async dispose() {},
} });
await cancelled.select(file("cancel.pdf")); const job = cancelled.recognize("eng", { mode: "all" }); await cancelled.cancel(); await job;
assert(cancelled.getState().phase === "cancelled" && cancelled.getState().pages.length === 0, "cancellation suppresses stale results");

let attempts = 0;
const retry = createPdfToTextController({ inspect: async () => ({ pageCount: 1 }), service: {
async recognizeDocument() { attempts += 1; if (attempts === 1) throw Object.assign(new Error("failed"), { code: "PDF_OCR_FAILED" }); return { pages: [{ pageNumber: 1, text: "recovered" }] }; },
async cancel() { return false; }, async dispose() {},
} });
await retry.select(file("retry.pdf")); await retry.recognize("eng", { mode: "all" }); await retry.recognize("eng", { mode: "all" });
assert(retry.getState().phase === "success", "retry recovers after a controlled failure");

let releaseOld;
const replacement = createPdfToTextController({ service: { async recognizeDocument() {}, async cancel() { return false; }, async dispose() {} }, inspect: async (source) => source.name === "old.pdf" ? new Promise((resolve) => { releaseOld = () => resolve({ pageCount: 9 }); }) : { pageCount: 2 } });
const oldJob = replacement.select(file("old.pdf")); await replacement.select(file("new.pdf")); releaseOld(); await oldJob;
assert(replacement.getState().source.file.name === "new.pdf", "source replacement rejects stale preparation");

let clipboard = "";
const text = formatPdfOcrText(happy.getState().pages);
await copyText(text, { navigatorObject: { clipboard: { writeText: async (value) => { clipboard = value; } } } });
assert(clipboard.includes("--- Page 1 ---") && clipboard.includes("--- Page 2 ---"), "copy-all preserves page boundaries");
let blob; const anchor = { click() { this.clicked = true; }, remove() {} };
downloadPdfText(happy.getState().pages, "multi.pdf", { documentObject: { body: { append() {} }, createElement: () => anchor }, urlObject: { createObjectURL(value) { blob = value; return "blob:test"; }, revokeObjectURL() {} }, schedule() {} });
assert(anchor.download === "multi.txt" && (await blob.text()) === text, "TXT export preserves filename and UTF-8 content");
assert(["en", "ko", "ja", "es", "de", "fr"].every((language) => translations[language].pdfToText.title && translations[language].pdfToText.result.copyAll), "all six locale catalogs load without raw keys");
show(); status.textContent = `${checks.length} browser checks passed.`; document.body.dataset.state = "passed";
} catch (error) {
status.textContent = error instanceof Error ? error.message : String(error); document.body.dataset.state = "failed"; console.error(error);
}
4 changes: 2 additions & 2 deletions tests/category-availability.test.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -84,8 +84,8 @@ assert.match(privacyHtml, /data-i18n="privacyHub\.pdfDescription"/);
assertPublicRoutesExist("/privacy/", privacyRoutes);

const pdfList = categoryList(read("tools/pdf/index.html"));
assert.equal(linkedRoutes(pdfList).length, 6, "Every PDF production card must remain linked");
assert.equal((pdfList.match(/status--available/g) || []).length, 6);
assert.equal(linkedRoutes(pdfList).length, 7, "Every PDF production card must remain linked");
assert.equal((pdfList.match(/status--available/g) || []).length, 7);

for (const category of ["pdf", "image", "scan"]) {
const html = read(`tools/${category}/index.html`);
Expand Down
4 changes: 2 additions & 2 deletions tests/i18n-quality.test.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -46,7 +46,7 @@ function placeholders(value) {
function testCatalogParityAndQuality() {
assert.deepEqual([...Object.keys(translations)], [...languageNames.keys()]);
const english = flatten(translations.en);
assert.equal(english.size, 817);
assert.equal(english.size, 887);

for (const [language, catalog] of Object.entries(translations)) {
const flattened = flatten(catalog);
Expand Down Expand Up @@ -78,7 +78,7 @@ function testResolutionDetectionAndPersistence() {

function testSelectorsAndDocumentTranslation() {
const pages = canonicalPages.map(({ source }) => path.join(root, source));
assert.equal(pages.length, 18, "Every canonical page comes from the route manifest");
assert.equal(pages.length, 19, "Every canonical page comes from the route manifest");
for (const file of pages) {
const html = fs.readFileSync(file, "utf8");
const select = html.match(/<select[^>]*data-language-select[^>]*>([\s\S]*?)<\/select>/)?.[1];
Expand Down
2 changes: 1 addition & 1 deletion tests/ocr-foundation.test.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -210,7 +210,7 @@ const publicOcrReferences = listAbsoluteFiles(path.join(root, "tools"))
.filter((file) => file.endsWith(".html"))
.map((file) => path.relative(root, file).replaceAll("\\", "/"))
.filter((relative) => read(relative).includes("assets/vendor/tesseract"));
assert.deepEqual(publicOcrReferences, ["tools/image/to-text/index.html"], "OCR runtime must stay lazy to its public route");
assert.deepEqual(publicOcrReferences, ["tools/image/to-text/index.html", "tools/pdf/to-text/index.html"], "OCR runtime must stay lazy to public OCR routes");
for (const required of ["engine/tesseract.min.js", "worker/worker.min.js", "lang/eng.traineddata.gz", "lang/kor.traineddata.gz"]) {
assert.ok(manifest.assets[required], `missing ${required}`);
}
Expand Down
98 changes: 98 additions & 0 deletions tests/pdf-to-text.test.mjs
Original file line number Diff line number Diff line change
@@ -0,0 +1,98 @@
import assert from "node:assert/strict";
import fs from "node:fs";
import path from "node:path";
import { fileURLToPath } from "node:url";
import { createPdfToTextController } from "../.ts-build/tools/pdf/to-text/controller.js";
import { copyText, downloadPdfText, formatPdfOcrText, pdfTextFilename } from "../.ts-build/tools/pdf/to-text/output.js";
import { canonicalPages, legacyRedirects } from "../scripts/site-routes.mjs";
import { translations } from "../js/i18n.js";

const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..");
const read = (file) => fs.readFileSync(path.join(root, file), "utf8");
const html = read("tools/pdf/to-text/index.html");

assert.equal(canonicalPages.length, 19);
assert.equal(legacyRedirects.length, 17);
assert.ok(canonicalPages.some(({ route }) => route === "/pdf/to-text/"));
assert.ok(!legacyRedirects.some(({ from }) => from === "/tools/pdf/to-text/"));
assert.match(html, /data-page="pdfToText"/);
assert.match(html, /href="https:\/\/tools\.securetools\.app\/pdf\/to-text\/"/);
assert.match(html, /id="pages-all"[\s\S]*id="pages-selected"[\s\S]*id="page-range"/);
assert.match(html, /value="eng"[\s\S]*value="kor"[\s\S]*value="eng\+kor"/);
assert.match(html, /connect-src 'none'/);
assert.doesNotMatch(html, /https?:\/\/(?!tools\.securetools\.app|github\.com)/);
for (const language of ["en", "ko", "ja", "es", "de", "fr"]) {
assert.ok(translations[language].metadata.pdfToText.title);
assert.ok(translations[language].pdfToText.result.copyAll);
assert.ok(translations[language].categories.pdf.toText);
}

assert.equal(formatPdfOcrText([{ pageNumber: 1, text: "One" }, { pageNumber: 3, text: "Three\n" }]), "--- Page 1 ---\nOne\n\n--- Page 3 ---\nThree\n");
assert.equal(pdfTextFilename("private/report.pdf"), "private_report.txt");

let resolveFirst;
const first = new Promise((resolve) => { resolveFirst = resolve; });
let request;
const states = [];
const service = {
async recognizeDocument(value) { request = value; await first; value.onPageResult?.({ pageNumber: 1, text: "stale" }); return { pages: [{ pageNumber: 1, text: "stale" }] }; },
async cancel() { resolveFirst?.(); return true; },
async dispose() {},
};
const controller = createPdfToTextController({ service, inspect: async () => ({ pageCount: 2 }), onChange: (state) => states.push(state.phase) });
const file = Object.assign(new Blob(["%PDF"], { type: "application/pdf" }), { name: "sample.pdf", lastModified: 0 });
await controller.select(file);
const running = controller.recognize("eng", { mode: "all" });
await controller.cancel();
await running;
assert.equal(controller.getState().phase, "cancelled");
assert.deepEqual(controller.getState().pages, []);
assert.equal(request.selection.mode, "all");
assert.ok(states.includes("recognizing"));

let copied = "";
await copyText(formatPdfOcrText([{ pageNumber: 2, text: "editable" }]), { navigatorObject: { clipboard: { writeText: async (text) => { copied = text; } } } });
assert.equal(copied, "--- Page 2 ---\neditable\n");
let downloaded;
const anchor = { click() { this.clicked = true; }, remove() {} };
downloadPdfText([{ pageNumber: 2, text: "editable" }], "unsafe:name.pdf", {
documentObject: { body: { append() {} }, createElement: () => anchor },
urlObject: { createObjectURL(blob) { downloaded = blob; return "blob:test"; }, revokeObjectURL() {} }, schedule() {},
});
assert.equal(anchor.download, "unsafe_name.txt");
assert.equal(anchor.clicked, true);
assert.equal(await downloaded.text(), "--- Page 2 ---\neditable\n");

let attempts = 0;
const retryController = createPdfToTextController({
inspect: async () => ({ pageCount: 1 }),
service: {
async recognizeDocument(value) {
attempts += 1;
if (attempts === 1) throw Object.assign(new Error("failed"), { code: "PDF_OCR_FAILED" });
const page = { pageNumber: 1, text: "retry worked" }; value.onPageResult?.(page);
return { pages: [page] };
},
async cancel() { return false; }, async dispose() {},
},
});
await retryController.select(file);
await retryController.recognize("eng", { mode: "all" });
assert.equal(retryController.getState().phase, "error");
await retryController.recognize("eng", { mode: "all" });
assert.equal(retryController.getState().phase, "success");
assert.equal(retryController.getState().pages[0].text, "retry worked");

let releaseOld;
const replacementController = createPdfToTextController({
service: { async recognizeDocument() { throw new Error("unused"); }, async cancel() { return false; }, async dispose() {} },
inspect: async (source) => source.name === "old.pdf" ? new Promise((resolve) => { releaseOld = () => resolve({ pageCount: 9 }); }) : { pageCount: 2 },
});
const oldFile = Object.assign(new Blob(["%PDF"], { type: "application/pdf" }), { name: "old.pdf", lastModified: 0 });
const newFile = Object.assign(new Blob(["%PDF"], { type: "application/pdf" }), { name: "new.pdf", lastModified: 0 });
const oldSelection = replacementController.select(oldFile);
await replacementController.select(newFile);
releaseOld(); await oldSelection;
assert.equal(replacementController.getState().source.file.name, "new.pdf");
assert.equal(replacementController.getState().source.pageCount, 2);
console.log("PDF to Text route, localization, output, and stale-job cancellation checks passed.");
1 change: 1 addition & 0 deletions tests/run-all.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -32,6 +32,7 @@ for (const test of [
"tests/image-metadata.test.mjs",
"tests/ocr-foundation.test.mjs",
"tests/pdf-ocr-foundation.test.mjs",
"tests/pdf-to-text.test.mjs",
"tests/image-to-text.test.mjs",
"tests/typescript-foundation.test.mjs",
"tests/category-availability.test.mjs",
Expand Down
2 changes: 1 addition & 1 deletion tests/seo-foundation.test.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -15,7 +15,7 @@ const indexableRoutes = new Map(canonicalPages.map(({ source, route }) => [sourc

const excludedRoutes = ["404.html", "tools/image-to-pdf/index.html"];
const allHtmlRoutes = [...indexableRoutes.keys(), ...excludedRoutes];
assert.equal(indexableRoutes.size, 18, "all canonical pages come from the route manifest");
assert.equal(indexableRoutes.size, 19, "all canonical pages come from the route manifest");
const shareImagePath = "assets/images/og-image.png";
const shareImageUrl = `${origin}/${shareImagePath}`;
const iconLinks = new Map([
Expand Down
1 change: 1 addition & 0 deletions tests/serve-ocr-smoke.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -52,5 +52,6 @@ const server = http.createServer((request, response) => {
server.listen(port, "127.0.0.1", () => {
console.log(`OCR browser smoke: http://127.0.0.1:${port}/tests/browser/ocr-smoke.html`);
console.log(`PDF OCR browser smoke: http://127.0.0.1:${port}/tests/browser/pdf-ocr-smoke.html`);
console.log(`PDF to Text browser smoke: http://127.0.0.1:${port}/tests/browser/pdf-to-text-smoke.html`);
console.log(`Image to Text UI QA: http://127.0.0.1:${port}/image/to-text/`);
});
Loading
Loading