Files
curriculum-project-hub/hub/src/capability/docmindClient.ts
T
2026-07-18 17:26:11 +08:00

232 lines
8.5 KiB
TypeScript

/**
* ADR-0027: Alibaba Cloud Document Mind (docmind) client.
*
* Uses the official @alicloud/docmind-api20220711 SDK to call the document
* parsing (large model version) API. The API is asynchronous:
* 1. SubmitDocParserJobAdvance — upload local file as a stream, get job id
* 2. QueryDocParserStatus — poll until data.status === "success";
* the markdown output URL is returned in outputFormatResult
* 3. Download markdown from OSS, extract and download referenced images
*
* Pricing (2025-07, aliyun docmind):
* - 图文文档基础链路: 0.02元/页
* - 图文文档增强链路 (含公式 LaTeX): 0.04元/页
* - 视频: 0.002元/秒
* - 音频: 0.00035元/秒
*/
import $DocmindClient, {
SubmitDocParserJobAdvanceRequest,
QueryDocParserStatusRequest,
} from "@alicloud/docmind-api20220711";
import { RuntimeOptions } from "@alicloud/tea-util";
import { createReadStream } from "node:fs";
import { basename } from "node:path";
import type { CapabilitySecretPayload } from "./types.js";
/** A single extracted image downloaded from the markdown's OSS image URLs. */
export interface DocmindExtractedImage {
readonly filename: string;
readonly data: Uint8Array;
}
/** The structured result of parsing one document. */
export interface DocmindParseResult {
readonly markdown: string;
readonly images: readonly DocmindExtractedImage[];
readonly pageCount: number;
readonly costUsd: number | null;
readonly requestId: string | null;
}
export interface DocmindParseOptions {
readonly inputFilePath: string;
}
export interface CapabilityProviderClient {
parse(credential: CapabilitySecretPayload, options: DocmindParseOptions): Promise<DocmindParseResult>;
}
export class DocmindClientError extends Error {
constructor(
message: string,
readonly code: "docmind_unreachable" | "docmind_rejected" | "docmind_invalid_response" | "docmind_no_output" | "docmind_timeout",
readonly upstreamStatus?: number,
) {
super(message);
this.name = "DocmindClientError";
}
}
/** Cost per page in USD (0.04 CNY/page ≈ 0.0056 USD). */
const COST_PER_PAGE_USD = 0.0056;
const POLL_INTERVAL_MS = 10_000;
const POLL_TIMEOUT_MS = 5 * 60_000;
type DocmindConfig = ConstructorParameters<typeof $DocmindClient.default>[0];
export class AliyunDocmindClient implements CapabilityProviderClient {
async parse(credential: CapabilitySecretPayload, options: DocmindParseOptions): Promise<DocmindParseResult> {
const config: DocmindConfig = {
endpoint: credential.endpoint,
accessKeyId: credential.accessKeyId,
accessKeySecret: credential.accessKeySecret,
type: "access_key",
regionId: "cn-hangzhou",
} as DocmindConfig;
const client = new $DocmindClient.default(config);
// 1. Submit job with local file as a ReadStream (not a Buffer — the SDK
// serializes Buffers as JSON {type:"Buffer",data:[...]} which the API
// can't read; a Stream is uploaded as multipart form data).
const fileName = basename(options.inputFilePath);
const fileStream = createReadStream(options.inputFilePath);
const advanceRequest = new SubmitDocParserJobAdvanceRequest({
fileUrlObject: fileStream,
fileName,
outputFormat: ["markdown"],
formulaEnhancement: true,
});
const runtime = new RuntimeOptions({});
let submitResponse;
try {
submitResponse = await client.submitDocParserJobAdvance(advanceRequest, runtime);
} catch (e) {
throw new DocmindClientError(
e instanceof Error ? e.message : String(e),
"docmind_unreachable",
);
}
const jobId = submitResponse.body?.data?.id;
if (jobId === undefined || jobId === null || jobId === "") {
throw new DocmindClientError("docmind submit returned no job id", "docmind_invalid_response");
}
// 2. Poll QueryDocParserStatus until data.status === "success" or "fail".
const deadline = Date.now() + POLL_TIMEOUT_MS;
let status = "";
let markdownUrl: string | null = null;
let pageCount = 0;
while (Date.now() < deadline) {
await sleep(POLL_INTERVAL_MS);
const statusReq = new QueryDocParserStatusRequest({ id: jobId });
const statusResponse = await client.queryDocParserStatus(statusReq);
const body = statusResponse.body as {
data?: {
status?: string;
pageCountEstimate?: number;
outputFormatResult?: Array<{ outputFileUrl?: string; outputType?: string }>;
};
};
status = body.data?.status ?? "";
if (status === "success") {
const mdResult = body.data?.outputFormatResult?.find((r) => r.outputType === "markdown");
markdownUrl = mdResult?.outputFileUrl ?? null;
pageCount = body.data?.pageCountEstimate ?? 0;
break;
}
if (status === "fail") {
throw new DocmindClientError(`docmind job ${jobId} failed`, "docmind_rejected");
}
}
if (status !== "success") {
throw new DocmindClientError(`docmind job ${jobId} timed out (status: ${status})`, "docmind_timeout");
}
if (markdownUrl === null) {
throw new DocmindClientError("docmind returned no markdown output URL", "docmind_no_output");
}
// 3. Download the markdown file from OSS.
let markdown: string;
try {
const mdResp = await fetch(markdownUrl);
if (!mdResp.ok) {
throw new DocmindClientError(`failed to download markdown: HTTP ${mdResp.status}`, "docmind_unreachable");
}
markdown = await mdResp.text();
} catch (e) {
if (e instanceof DocmindClientError) throw e;
throw new DocmindClientError(
e instanceof Error ? e.message : String(e),
"docmind_unreachable",
);
}
if (markdown === "") {
throw new DocmindClientError("docmind returned empty markdown", "docmind_no_output");
}
// 4. Download images referenced in the markdown (OSS URLs).
// Markdown contains ![filename](http://...oss.../image.png?...) entries.
// We rewrite them to local relative paths and download the images.
const images: DocmindExtractedImage[] = [];
const imageRefRegex = /!\[([^\]]*)\]\((https?:\/\/[^)]+)\)/g;
const localMarkdown = markdown.replace(imageRefRegex, (match, altText: string, url: string) => {
const idx = images.findIndex((img) => img.filename === extractFilename(altText, url, images.length));
if (idx >= 0) {
const img = images[idx]!;
return `![${altText}](${img.filename})`;
}
return match;
});
// Collect all image URLs first, then download.
const imageUrls: Array<{ url: string; altText: string }> = [];
let match: RegExpExecArray | null;
const collectRegex = /!\[([^\]]*)\]\((https?:\/\/[^)]+)\)/g;
while ((match = collectRegex.exec(markdown)) !== null) {
imageUrls.push({ url: match[2]!, altText: match[1]! });
}
for (let i = 0; i < imageUrls.length; i++) {
const { url, altText } = imageUrls[i]!;
const filename = extractFilename(altText, url, i);
try {
const imgResp = await fetch(url);
if (!imgResp.ok) continue;
const data = new Uint8Array(await imgResp.arrayBuffer());
images.push({ filename, data });
} catch {
// Best-effort: skip images that fail to download.
}
}
// Rewrite markdown with local image paths.
let finalMarkdown = markdown;
let imageIdx = 0;
finalMarkdown = finalMarkdown.replace(imageRefRegex, (match, altText: string, _url: string) => {
if (imageIdx < images.length) {
const img = images[imageIdx]!;
imageIdx++;
return `![${altText}](${img.filename})`;
}
return match;
});
if (pageCount === 0) {
pageCount = Math.max(1, images.length);
}
const costUsd = (pageCount > 0 ? pageCount : 1) * COST_PER_PAGE_USD;
return { markdown: finalMarkdown, images, pageCount, costUsd, requestId: jobId };
}
}
function sleep(ms: number): Promise<void> {
return new Promise((resolve) => {
setTimeout(resolve, ms);
});
}
/** Derive a clean filename from the alt text or URL. */
function extractFilename(altText: string, url: string, index: number): string {
// Try alt text first (docmind often puts the original filename).
if (altText !== "" && altText.length < 100) {
const cleaned = altText.replace(/[^a-zA-Z0-9._-]/g, "_");
if (cleaned.length > 0) return cleaned;
}
// Fall back to URL path.
const urlPath = new URL(url).pathname;
const base = basename(urlPath);
if (base !== "" && base !== "/") return base;
return `image_${index + 1}.png`;
}