chore: 1.0.0 release

This commit is contained in:
binaryify
2026-08-05 16:25:24 +08:00
commit 5b0d45bf1d
100 changed files with 11034 additions and 0 deletions
@@ -0,0 +1,175 @@
<!doctype html>
<html lang="zh-CN">
<head>
<meta charset="utf-8" />
<meta name="viewport" content="width=device-width, initial-scale=1" />
<title>Open Kimi PPT Local Export</title>
<style>
html,
body,
iframe {
width: 100%;
height: 100%;
margin: 0;
border: 0;
}
#status {
position: fixed;
z-index: 10;
top: 8px;
left: 50%;
transform: translateX(-50%);
padding: 4px 10px;
border-radius: 6px;
background: rgb(0 0 0 / 72%);
color: white;
font: 12px/1.5 system-ui, sans-serif;
pointer-events: none;
}
</style>
</head>
<body>
<div id="status">starting</div>
<iframe id="editor" allow="fullscreen; clipboard-read; clipboard-write"></iframe>
<script type="module">
import {
i as connect,
r as WindowMessenger,
} from "https://statics.moonshot.cn/neo-design/assets/penpal-C4NjirZE.js";
const status = document.querySelector("#status");
const frame = document.querySelector("#editor");
const editorOrigin = "https://www.kimi.com";
const setStatus = (value) => {
status.textContent = value;
document.documentElement.dataset.deckStatus = value;
};
const normalizePath = (value) => {
if (typeof value !== "string") return "";
if (/^(?:data:image\/|https?:\/\/|blob:)/i.test(value)) return value;
let candidate = value.replace(/^file:\/\/+/, "").replaceAll("\\", "/");
try {
candidate = decodeURIComponent(candidate);
} catch {
// Preserve literal percent signs.
}
const parts = [];
for (const part of candidate.split("/")) {
if (!part || part === ".") continue;
if (part === "..") parts.pop();
else parts.push(part);
}
return parts.join("/");
};
const resolveImage = (requestedPath, imageMap) => {
if (/^(?:data:image\/|https?:\/\/|blob:)/i.test(requestedPath || "")) {
return requestedPath;
}
const normalized = normalizePath(requestedPath);
if (imageMap[normalized]) return imageMap[normalized];
const suffix = Object.keys(imageMap).find(
(path) => normalized.endsWith(`/${path}`) || path.endsWith(`/${normalized}`),
);
return suffix ? imageMap[suffix] : "";
};
try {
const payload = await fetch("./payload.json", { cache: "no-store" }).then((response) => {
if (!response.ok) throw new Error(`payload HTTP ${response.status}`);
return response.json();
});
const query = new URLSearchParams({
sdkMode: "ppt-editor",
pptPlatform: "open-kimi-ppt-local-export",
functional: JSON.stringify({
fullscreen: false,
present: false,
export: true,
close: false,
annotation: false,
feedback: false,
share: false,
versionHistory: false,
}),
sdkSaveMode: "external",
sdkImageMode: "external",
});
frame.addEventListener(
"load",
async () => {
setStatus("iframe-loaded");
try {
const messenger = new WindowMessenger({
remoteWindow: frame.contentWindow,
allowedOrigins: [editorOrigin],
});
const connection = connect({
messenger,
methods: {
close() {},
reenter() {},
toggleFullScreen(value) {
return value;
},
showFeedback() {},
sendPrompt() {},
showMessage() {},
hideMessage() {},
onSave(savePayload) {
return {
fileContent: savePayload?.fileContent,
lastModifiedTime: Date.now(),
};
},
getImages(imagePayload = {}) {
const paths = Array.isArray(imagePayload.filePath) ? imagePayload.filePath : [];
return paths.map((path) => resolveImage(path, payload.imageMap || {}));
},
setAnnotationMode() {},
setAnnotationCurrentPage() {},
upsertAnnotation() {},
removeAnnotation() {},
clearAnnotations() {},
},
});
const remote = await connection.promise;
window.exportRemote = remote;
await remote.setSlideConfig({
editable: true,
locale: "zh-CN",
theme: "light",
slideId: payload.id,
});
await remote.setPPTD(payload.id, {
pptdContent: payload.manifestContent,
pages: payload.pages,
pptdPath: payload.manifestPath,
basePath: "",
isCreate: true,
});
await remote.setEditable(true);
window.exportSlideStatus = await remote.getSlideStatus();
setStatus("ready");
} catch (error) {
window.exportHostError = String(error?.stack || error);
setStatus("error");
}
},
{ once: true },
);
frame.src = `${editorOrigin}/neo-ppt/?${query}`;
} catch (error) {
window.exportHostError = String(error?.stack || error);
setStatus("error");
}
</script>
</body>
</html>
@@ -0,0 +1,427 @@
#!/usr/bin/env python3
"""Export a PPTD project as page images through Kimi's public editor for visual QA.
Reuses the same localhost SDK host and agent-browser flow as export_pptx.py, but
chooses 图片 in the export dialog, captures the images ZIP, unzips it, and stitches
all pages into a single overview image that a multimodal model can review.
"""
from __future__ import annotations
import argparse
import json
import math
import os
import re
import shutil
import subprocess
import sys
import tempfile
import time
import uuid
import zipfile
from pathlib import Path
from typing import Any, Dict, List, Optional, Sequence, Tuple
from export_pptx import (
HOST_TEMPLATE,
BrowserSession,
ExportError,
build_payload,
ensure_agent_browser,
find_download,
find_manifest,
log,
ref_by_name,
serve,
wait_for_export_dialog,
)
IMAGE_SUFFIXES = {".png", ".jpg", ".jpeg", ".webp"}
OVERVIEW_COLUMNS = 3
OVERVIEW_THUMB_WIDTH = 640
OVERVIEW_LABEL_HEIGHT = 32
OVERVIEW_GAP = 12
def ensure_pillow() -> Tuple[Any, Any, Any]:
try:
from PIL import Image, ImageDraw, ImageFont
return Image, ImageDraw, ImageFont
except ImportError:
log("Pillow is required for stitching; installing pillow with pip --user")
process = subprocess.run(
[sys.executable, "-m", "pip", "install", "--user", "pillow"],
text=True,
stdout=subprocess.PIPE,
stderr=subprocess.STDOUT,
timeout=300,
)
if process.returncode != 0:
raise ExportError(f"failed to install Pillow:\n{process.stdout[-2000:]}")
from PIL import Image, ImageDraw, ImageFont
return Image, ImageDraw, ImageFont
def is_image_zip(path: Path) -> bool:
if not path.is_file() or path.name.endswith(".crdownload"):
return False
try:
with zipfile.ZipFile(path) as archive:
return any(
Path(name).suffix.lower() in IMAGE_SUFFIXES
for name in archive.namelist()
)
except (OSError, zipfile.BadZipFile):
return False
def page_sort_key(path: Path) -> Tuple[int, str]:
match = re.match(r"(\d+)", path.stem)
return (int(match.group(1)) if match else sys.maxsize, path.name)
def unzip_images(archive_path: Path, pages_dir: Path) -> List[Path]:
pages_dir.mkdir(parents=True, exist_ok=True)
images: List[Path] = []
with zipfile.ZipFile(archive_path) as archive:
for info in archive.infolist():
if info.is_dir() or Path(info.filename).suffix.lower() not in IMAGE_SUFFIXES:
continue
name = Path(info.filename).name
if not name:
continue
target = pages_dir / name
with archive.open(info) as source, target.open("wb") as out:
shutil.copyfileobj(source, out)
images.append(target)
images.sort(key=page_sort_key)
if not images:
raise ExportError(f"no page images found in: {archive_path}")
return images
def label_font(image_font: Any) -> Any:
try:
return image_font.load_default(size=18)
except TypeError: # older Pillow without the size argument
return image_font.load_default()
def stitch_overview(
images: Sequence[Path],
output: Path,
image_cls: Any,
draw_cls: Any,
image_font: Any,
) -> Path:
thumbs: List[Tuple[str, Any]] = []
for index, path in enumerate(images, start=1):
with image_cls.open(path) as opened:
frame = opened.convert("RGB")
ratio = OVERVIEW_THUMB_WIDTH / frame.width
thumb = frame.resize(
(OVERVIEW_THUMB_WIDTH, max(1, round(frame.height * ratio)))
)
thumbs.append((f"P{index}", thumb))
columns = OVERVIEW_COLUMNS
rows = math.ceil(len(thumbs) / columns)
cell_height = OVERVIEW_LABEL_HEIGHT + max(thumb.height for _, thumb in thumbs)
width = columns * OVERVIEW_THUMB_WIDTH + (columns + 1) * OVERVIEW_GAP
height = rows * cell_height + (rows + 1) * OVERVIEW_GAP
overview = image_cls.new("RGB", (width, height), "#e5e7eb")
draw = draw_cls.Draw(overview)
font = label_font(image_font)
for position, (label, thumb) in enumerate(thumbs):
column = position % columns
row = position // columns
x = OVERVIEW_GAP + column * (OVERVIEW_THUMB_WIDTH + OVERVIEW_GAP)
y = OVERVIEW_GAP + row * (cell_height + OVERVIEW_GAP)
draw.rectangle(
(x, y, x + OVERVIEW_THUMB_WIDTH, y + OVERVIEW_LABEL_HEIGHT - 4),
fill="#111827",
)
draw.text((x + 8, y + 5), label, fill="#ffffff", font=font)
overview.paste(thumb, (x, y + OVERVIEW_LABEL_HEIGHT))
overview.save(output, "JPEG", quality=85)
return output
OOPIF_URL_HINT = "kimi.com/neo-ppt"
# The export dialog's 图片 format option is a plain <div class="radio-group-item">
# without an ARIA role, so agent-browser's interactive snapshot never exposes it
# and cross-origin iframe rules block page-level eval. Clicking it requires CDP.
IMAGE_FORMAT_CLICK_JS = """
(() => {
const items = [...document.querySelectorAll('.radio-group-item')];
const pool = items.length
? items
: [...document.querySelectorAll('div,span,label,button')].filter(
(el) => el.children.length === 0
);
const target = pool.find((el) => el.textContent.trim() === '图片');
if (!target) return null;
target.click();
return 'clicked';
})()
""".strip()
ACTIVE_FORMAT_JS = """
(() => {
const active = document.querySelector('.radio-group-item.active');
return active ? active.textContent.trim() : null;
})()
""".strip()
def ensure_websocket() -> Any:
try:
import websocket
return websocket
except ImportError:
log("websocket-client is required for dialog automation; installing with pip --user")
process = subprocess.run(
[sys.executable, "-m", "pip", "install", "--user", "websocket-client"],
text=True,
stdout=subprocess.PIPE,
stderr=subprocess.STDOUT,
timeout=300,
)
if process.returncode != 0:
raise ExportError(f"failed to install websocket-client:\n{process.stdout[-2000:]}")
import websocket
return websocket
def browser_cdp_url(browser: BrowserSession) -> str:
process = browser.run(["get", "cdp-url"], timeout=30)
match = re.search(r"ws://\S+", process.stdout)
if not match:
raise ExportError(
f"could not determine the browser CDP URL:\n{process.stdout[-500:]}"
)
return match.group(0)
def evaluate_in_iframe(cdp_url: str, url_hint: str, expression: str) -> Any:
websocket = ensure_websocket()
def call(socket: Any, request_id: int, method: str, params: Dict[str, Any]) -> Dict[str, Any]:
socket.send(json.dumps({"id": request_id, "method": method, "params": params}))
while True:
message = json.loads(socket.recv())
if message.get("id") != request_id:
continue
if "error" in message:
raise ExportError(f"CDP {method} failed: {message['error']}")
return message.get("result", {})
# websocket-client honors http_proxy env vars; the CDP endpoint is local,
# so strip proxy settings instead of tunneling localhost through the proxy.
proxy_env = ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY", "all_proxy", "ALL_PROXY")
saved_proxy = {name: os.environ.pop(name) for name in proxy_env if name in os.environ}
try:
socket = websocket.create_connection(cdp_url, timeout=30, suppress_origin=True)
finally:
os.environ.update(saved_proxy)
try:
targets = call(socket, 1, "Target.getTargets", {}).get("targetInfos", [])
target = next(
(item for item in targets if url_hint in str(item.get("url", ""))), None
)
if target is None:
visible = ", ".join(
f"{item.get('type')}:{str(item.get('url', ''))[:80]}" for item in targets
)
raise ExportError(
f"no browser target matches {url_hint!r}; observed: {visible}"
)
attached = call(
socket,
2,
"Target.attachToTarget",
{"targetId": target["targetId"], "flatten": True},
)
session_id = attached["sessionId"]
socket.send(
json.dumps(
{
"id": 3,
"sessionId": session_id,
"method": "Runtime.evaluate",
"params": {"expression": expression, "returnByValue": True},
}
)
)
while True:
message = json.loads(socket.recv())
if message.get("id") != 3:
continue
if "error" in message:
raise ExportError(f"CDP Runtime.evaluate failed: {message['error']}")
result = message.get("result", {})
if result.get("exceptionDetails"):
details = result["exceptionDetails"]
raise ExportError(f"iframe script failed: {details.get('text')}")
return result.get("result", {}).get("value")
finally:
socket.close()
def select_image_format(browser: BrowserSession) -> None:
cdp_url = browser_cdp_url(browser)
value = evaluate_in_iframe(cdp_url, OOPIF_URL_HINT, IMAGE_FORMAT_CLICK_JS)
if value != "clicked":
raise ExportError("could not find the 图片 option in the export dialog")
deadline = time.monotonic() + 10
while time.monotonic() < deadline:
active = evaluate_in_iframe(cdp_url, OOPIF_URL_HINT, ACTIVE_FORMAT_JS)
if active == "图片":
return
time.sleep(0.3)
raise ExportError(f"image format was not activated; active option: {active!r}")
def export_images(
source: Path,
output: Path,
keep_download: bool = False,
force: bool = False,
) -> Dict[str, Any]:
manifest = find_manifest(source)
payload = build_payload(manifest)
output = output.expanduser().resolve()
if output.exists() and any(output.iterdir()) and not force:
raise ExportError(
f"output directory already exists (pass --force to replace it): {output}"
)
agent_browser = ensure_agent_browser()
image_cls, draw_cls, image_font = ensure_pillow()
log(f"manifest: {manifest}")
with tempfile.TemporaryDirectory(prefix="open-kimi-ppt-images-") as temp_name:
temp_dir = Path(temp_name)
download_dir = temp_dir / "downloads"
download_dir.mkdir()
shutil.copy2(HOST_TEMPLATE, temp_dir / HOST_TEMPLATE.name)
(temp_dir / "payload.json").write_text(
json.dumps(payload, ensure_ascii=False), encoding="utf-8"
)
server, thread, url = serve(temp_dir)
session = f"open-kimi-ppt-images-{os.getpid()}-{uuid.uuid4().hex[:8]}"
browser = BrowserSession(agent_browser, session, temp_dir, download_dir)
try:
log("opening the public Kimi slide editor")
browser.open(url)
browser.run(
[
"wait",
"--fn",
'document.documentElement.dataset.deckStatus === "ready"',
],
timeout=120,
)
browser.run(["set", "viewport", "1280", "720"])
snapshot = browser.snapshot()
export_ref = ref_by_name(snapshot, "导出", "button")
browser.run(["click", f"@{export_ref}"])
dialog = wait_for_export_dialog(browser)
select_image_format(browser)
dialog = wait_for_export_dialog(browser)
download_ref = ref_by_name(dialog, "下载", "button")
log("rendering page images in the browser")
result = browser.run(
["download", f"@{download_ref}", str(temp_dir / "browser-output.zip")],
timeout=300,
check=False,
)
if result.returncode != 0:
log("download capture reported a timeout; checking browser output files")
downloaded = find_download(
(download_dir, temp_dir), timeout=240, accept=is_image_zip
)
finally:
browser.close()
server.shutdown()
server.server_close()
thread.join(timeout=2)
if output.exists():
shutil.rmtree(output)
output.mkdir(parents=True)
images = unzip_images(downloaded, output / "pages")
if keep_download:
shutil.copy2(downloaded, output / "browser-raw.zip")
overview = stitch_overview(
images, output / "overview.jpg", image_cls, draw_cls, image_font
)
page_paths = [entry["path"] for entry in payload["pages"]]
mapping = [
{
"index": index,
"image": f"pages/{path.name}",
"page": page_paths[index - 1] if index - 1 < len(page_paths) else None,
}
for index, path in enumerate(images, start=1)
]
return {
"pages": len(images),
"overview": str(overview),
"output": str(output),
"images": mapping,
}
def parse_args(argv: Optional[Sequence[str]] = None) -> argparse.Namespace:
parser = argparse.ArgumentParser(
description=(
"Export a PPTD project as page images via Kimi's public editor, unzip "
"them, and stitch an overview image for visual QA."
)
)
parser.add_argument("input", type=Path, help=".pptd manifest or project directory")
parser.add_argument(
"--output",
"-o",
type=Path,
help="output directory (default: <project>/.qa-images)",
)
parser.add_argument(
"--keep-browser-raw",
action="store_true",
help="also keep the downloaded images ZIP beside the overview",
)
parser.add_argument(
"--force",
action="store_true",
help="replace an existing output directory",
)
return parser.parse_args(argv)
def main(argv: Optional[Sequence[str]] = None) -> int:
args = parse_args(argv)
try:
manifest = find_manifest(args.input)
output = args.output or manifest.parent / ".qa-images"
summary = export_images(args.input, output, args.keep_browser_raw, args.force)
except (ExportError, OSError, subprocess.SubprocessError) as exc:
print(f"open-kimi-ppt image export failed: {exc}", file=sys.stderr)
return 1
print(json.dumps(summary, ensure_ascii=False, indent=2))
return 0
if __name__ == "__main__":
raise SystemExit(main())
+662
View File
@@ -0,0 +1,662 @@
#!/usr/bin/env python3
"""Export a PPTD project through Kimi's public browser-side PPTX writer.
The script uses a temporary localhost SDK host and agent-browser. It never uploads
the PPTD project as a document. Referenced remote resources may still be fetched by
the official editor. Local image files are exposed to the iframe as data URLs.
"""
from __future__ import annotations
import argparse
import base64
import json
import mimetypes
import os
import re
import shutil
import socket
import subprocess
import sys
import tempfile
import threading
import time
import uuid
import zipfile
import xml.etree.ElementTree as ET
from http.server import SimpleHTTPRequestHandler, ThreadingHTTPServer
from pathlib import Path
from typing import Any, Callable, Dict, Iterable, List, Optional, Sequence, Tuple
try:
import yaml
except ImportError as exc: # pragma: no cover - environment diagnostic
raise SystemExit(
"PyYAML is required. Install it with: python3 -m pip install --user pyyaml"
) from exc
SKILL_DIR = Path(__file__).resolve().parent.parent
HOST_TEMPLATE = Path(__file__).with_name("export_host.html")
IMAGE_MIME = {
".png": "image/png",
".jpg": "image/jpeg",
".jpeg": "image/jpeg",
".gif": "image/gif",
".svg": "image/svg+xml",
}
MAX_IMAGE_BYTES = 20 * 1024 * 1024
MAX_EMBEDDED_MEDIA_BYTES = 200 * 1024 * 1024
PPTX_CONTENT_TYPE = (
"application/vnd.openxmlformats-officedocument.presentationml.presentation.main+xml"
)
FADE_TRANSITION_XML = (
'<p:transition spd="fast" advClick="1"><p:fade/></p:transition>'
)
MIN_AGENT_BROWSER_VERSION = (0, 33, 2)
class ExportError(RuntimeError):
pass
class QuietHandler(SimpleHTTPRequestHandler):
def log_message(self, _format: str, *_args: Any) -> None:
return
def log(message: str) -> None:
print(f"[open-kimi-ppt] {message}", file=sys.stderr, flush=True)
def parse_version(output: str) -> Tuple[int, int, int]:
match = re.search(r"(\d+)\.(\d+)\.(\d+)\b", output)
if not match:
raise ExportError(f"could not parse agent-browser version from: {output.strip()}")
return tuple(int(part) for part in match.groups())
def read_agent_browser_version(executable: str) -> Tuple[int, int, int]:
process = subprocess.run(
[executable, "--version"],
text=True,
stdout=subprocess.PIPE,
stderr=subprocess.STDOUT,
timeout=30,
)
if process.returncode != 0:
raise ExportError(f"agent-browser --version failed:\n{process.stdout[-2000:]}")
return parse_version(process.stdout)
def ensure_agent_browser() -> str:
executable = shutil.which("agent-browser")
version = read_agent_browser_version(executable) if executable else None
if version is not None and version >= MIN_AGENT_BROWSER_VERSION:
log(f"agent-browser version: {'.'.join(map(str, version))}")
return executable
npm = shutil.which("npm")
if not npm:
reason = "not installed" if version is None else ".".join(map(str, version))
raise ExportError(
f"agent-browser {reason}; npm is required to install agent-browser@latest"
)
current = "not installed" if version is None else ".".join(map(str, version))
minimum = ".".join(map(str, MIN_AGENT_BROWSER_VERSION))
log(f"agent-browser {current} is below {minimum}; installing agent-browser@latest")
process = subprocess.run(
[npm, "install", "-g", "agent-browser@latest"],
text=True,
stdout=subprocess.PIPE,
stderr=subprocess.STDOUT,
timeout=300,
)
if process.returncode != 0:
raise ExportError(f"failed to install agent-browser@latest:\n{process.stdout[-4000:]}")
executable = shutil.which("agent-browser")
if not executable:
raise ExportError("agent-browser@latest installed, but executable is not on PATH")
version = read_agent_browser_version(executable)
if version < MIN_AGENT_BROWSER_VERSION:
raise ExportError(
"agent-browser@latest is still below the required version "
f"{minimum}: {'.'.join(map(str, version))}"
)
log(f"agent-browser upgraded to {'.'.join(map(str, version))}")
return executable
def find_manifest(source: Path) -> Path:
source = source.expanduser().resolve()
if source.is_file():
if source.suffix.lower() != ".pptd":
raise ExportError(f"input must be a .pptd file or project directory: {source}")
return source
if not source.is_dir():
raise ExportError(f"input does not exist: {source}")
manifests = sorted(source.rglob("*.pptd"))
if not manifests:
raise ExportError(f"no .pptd manifest found under: {source}")
if len(manifests) > 1:
choices = "\n ".join(str(path) for path in manifests[:20])
raise ExportError(
"multiple .pptd manifests found; pass one manifest explicitly:\n " + choices
)
return manifests[0]
def read_yaml_mapping(path: Path) -> Tuple[str, Dict[str, Any]]:
text = path.read_text(encoding="utf-8")
try:
value = yaml.safe_load(text)
except yaml.YAMLError as exc:
raise ExportError(f"invalid YAML in {path}: {exc}") from exc
if not isinstance(value, dict):
raise ExportError(f"expected a YAML mapping in {path}")
return text, value
def safe_project_path(root: Path, relative: str) -> Path:
if not isinstance(relative, str) or not relative.strip():
raise ExportError("page path must be a non-empty string")
candidate = (root / relative).resolve()
try:
candidate.relative_to(root)
except ValueError as exc:
raise ExportError(f"project path escapes the PPTD directory: {relative}") from exc
return candidate
def build_image_map(root: Path) -> Dict[str, str]:
image_map: Dict[str, str] = {}
total = 0
for path in sorted(root.rglob("*")):
if not path.is_file() or path.suffix.lower() not in IMAGE_MIME:
continue
size = path.stat().st_size
if size > MAX_IMAGE_BYTES:
log(f"skip local image over 20 MiB: {path.relative_to(root)}")
continue
if total + size > MAX_EMBEDDED_MEDIA_BYTES:
raise ExportError(
"local image payload exceeds 200 MiB; reduce media size or use remote URLs"
)
data = base64.b64encode(path.read_bytes()).decode("ascii")
rel = path.relative_to(root).as_posix()
image_map[rel] = f"data:{IMAGE_MIME[path.suffix.lower()]};base64,{data}"
total += size
if image_map:
log(f"prepared {len(image_map)} local image resource(s), {total} bytes")
return image_map
def build_payload(manifest: Path) -> Dict[str, Any]:
manifest_text, manifest_data = read_yaml_mapping(manifest)
if manifest_data.get("version") != "v2":
raise ExportError("local PPTX export currently requires PPTD version: v2")
page_paths = manifest_data.get("pages")
if not isinstance(page_paths, list) or not page_paths:
raise ExportError("PPTD manifest must contain a non-empty pages list")
root = manifest.parent.resolve()
pages: List[Dict[str, str]] = []
for entry in page_paths:
page_path = safe_project_path(root, entry)
if not page_path.is_file():
raise ExportError(f"missing page file: {entry}")
page_text, page_data = read_yaml_mapping(page_path)
if not isinstance(page_data.get("elements"), list):
raise ExportError(f"page elements must be an array: {entry}")
pages.append({"path": str(entry), "content": page_text})
title = str(manifest_data.get("title") or manifest.stem)
return {
"id": f"local-export-{uuid.uuid4().hex}",
"title": title,
"manifestPath": manifest.name,
"manifestContent": manifest_text,
"pages": pages,
"imageMap": build_image_map(root),
}
def json_result(output: str) -> Dict[str, Any]:
for line in reversed(output.splitlines()):
line = line.strip()
if not line.startswith("{"):
continue
try:
value = json.loads(line)
except json.JSONDecodeError:
continue
if isinstance(value, dict):
return value
raise ExportError(f"agent-browser returned no JSON object:\n{output[-2000:]}")
class BrowserSession:
def __init__(self, executable: str, session: str, cwd: Path, download_dir: Path):
self.executable = executable
self.session = session
self.cwd = cwd
self.download_dir = download_dir
self.env = os.environ.copy()
self.env.setdefault("AGENT_BROWSER_DEFAULT_TIMEOUT", "60000")
self.env.setdefault("AGENT_BROWSER_IDLE_TIMEOUT_MS", "180000")
def run(
self,
args: Sequence[str],
*,
timeout: int = 90,
check: bool = True,
) -> subprocess.CompletedProcess[str]:
command = [self.executable, "--session", self.session, *args]
process = subprocess.run(
command,
cwd=self.cwd,
env=self.env,
text=True,
stdout=subprocess.PIPE,
stderr=subprocess.STDOUT,
timeout=timeout,
)
if check and process.returncode != 0:
raise ExportError(
f"agent-browser command failed ({process.returncode}): "
f"{' '.join(args)}\n{process.stdout[-4000:]}"
)
return process
def open(self, url: str) -> None:
self.run(
["--download-path", str(self.download_dir), "open", url], timeout=90
)
def snapshot(self) -> Dict[str, Any]:
process = self.run(["snapshot", "-i", "-C", "--json"])
return json_result(process.stdout)
def close(self) -> None:
self.run(["close"], timeout=20, check=False)
def snapshot_data(snapshot: Dict[str, Any]) -> Dict[str, Any]:
data = snapshot.get("data")
if not isinstance(data, dict):
raise ExportError(f"invalid agent-browser snapshot: {snapshot}")
return data
def ref_by_name(snapshot: Dict[str, Any], name: str, role: Optional[str] = None) -> str:
refs = snapshot_data(snapshot).get("refs")
if not isinstance(refs, dict):
raise ExportError("snapshot contains no interactive refs")
matches = []
for ref, metadata in refs.items():
if not isinstance(metadata, dict) or metadata.get("name") != name:
continue
if role is not None and str(metadata.get("role", "")).lower() != role.lower():
continue
matches.append(ref)
if not matches:
raise ExportError(f"could not find {role or 'element'} named {name!r}")
return matches[-1]
def switch_state(snapshot: Dict[str, Any]) -> Optional[Tuple[str, bool, bool]]:
text = str(snapshot_data(snapshot).get("snapshot") or "")
match = re.search(r"switch \[(?P<attrs>[^\]]*?)ref=(?P<ref>e\d+)\]", text)
if not match:
return None
attrs = match.group("attrs")
return match.group("ref"), "checked=true" in attrs, "disabled" in attrs
def wait_for_export_dialog(browser: BrowserSession, timeout: float = 20.0) -> Dict[str, Any]:
deadline = time.monotonic() + timeout
last: Optional[Dict[str, Any]] = None
while time.monotonic() < deadline:
last = browser.snapshot()
try:
ref_by_name(last, "下载", "button")
return last
except ExportError:
time.sleep(0.35)
raise ExportError(f"export dialog did not become ready: {last}")
def is_pptx(path: Path) -> bool:
if not path.is_file() or path.name.endswith(".crdownload"):
return False
try:
with zipfile.ZipFile(path) as archive:
if "ppt/presentation.xml" not in archive.namelist():
return False
content_types = archive.read("[Content_Types].xml")
return PPTX_CONTENT_TYPE.encode("utf-8") in content_types
except (OSError, KeyError, zipfile.BadZipFile):
return False
def find_download(
search_roots: Iterable[Path],
timeout: float = 150.0,
accept: Callable[[Path], bool] = is_pptx,
) -> Path:
deadline = time.monotonic() + timeout
last_sizes: Dict[Path, int] = {}
stable: Dict[Path, int] = {}
while time.monotonic() < deadline:
candidates: List[Path] = []
for root in search_roots:
if not root.exists():
continue
candidates.extend(path for path in root.rglob("*") if path.is_file())
for path in sorted(candidates, key=lambda item: item.stat().st_mtime, reverse=True):
try:
size = path.stat().st_size
except OSError:
continue
if size == last_sizes.get(path) and size > 0:
stable[path] = stable.get(path, 0) + 1
else:
stable[path] = 0
last_sizes[path] = size
if stable[path] >= 1 and accept(path):
return path
time.sleep(0.5)
visible = "\n ".join(str(path) for path in last_sizes) or "(none)"
raise ExportError(f"timed out waiting for download; observed files:\n {visible}")
def replace_transition(slide_xml: bytes, transition: str) -> bytes:
text = slide_xml.decode("utf-8")
pattern = re.compile(
r"<p:transition\b[^>]*(?:/>|>.*?</p:transition>)", re.DOTALL
)
text = pattern.sub("", text)
if transition == "none":
return text.encode("utf-8")
# CT_Slide requires transition as a direct child after cSld/clrMapOvr and
# before timing/extLst. Searching for the first p:extLst is incorrect:
# shapes may contain their own nested extLst inside cSld, causing Office to
# ignore a transition inserted there.
color_map = re.search(
r"<p:clrMapOvr\b[^>]*(?:/>|>.*?</p:clrMapOvr>)", text, re.DOTALL
)
common_slide = re.search(
r"<p:cSld\b[^>]*(?:/>|>.*?</p:cSld>)", text, re.DOTALL
)
anchor = color_map or common_slide
if anchor is None:
raise ExportError("slide XML has no cSld/clrMapOvr insertion anchor")
position = anchor.end()
return (text[:position] + FADE_TRANSITION_XML + text[position:]).encode("utf-8")
def root_child_names(slide_xml: bytes) -> List[str]:
try:
root = ET.fromstring(slide_xml)
except ET.ParseError as exc:
raise ExportError(f"invalid slide XML: {exc}") from exc
return [child.tag.rsplit("}", 1)[-1] for child in root]
def has_direct_fade_transition(slide_xml: bytes) -> bool:
try:
root = ET.fromstring(slide_xml)
except ET.ParseError as exc:
raise ExportError(f"invalid slide XML: {exc}") from exc
transition = next(
(child for child in root if child.tag.rsplit("}", 1)[-1] == "transition"),
None,
)
if transition is None:
return False
return any(child.tag.rsplit("}", 1)[-1] == "fade" for child in transition)
def validate_transition_order(slide_xml: bytes, transition: str) -> None:
names = root_child_names(slide_xml)
transition_indexes = [index for index, name in enumerate(names) if name == "transition"]
if transition == "none":
if transition_indexes:
raise ExportError("transition=none left a root-level transition")
return
if len(transition_indexes) != 1 or not has_direct_fade_transition(slide_xml):
raise ExportError("slide does not contain exactly one root-level fade transition")
transition_index = transition_indexes[0]
for required_before in ("cSld", "clrMapOvr"):
if required_before in names and names.index(required_before) > transition_index:
raise ExportError(f"{required_before} appears after transition")
for required_after in ("timing", "extLst"):
if required_after in names and names.index(required_after) < transition_index:
raise ExportError(f"{required_after} appears before transition")
def patch_transitions(pptx: Path, transition: str) -> int:
temporary = pptx.with_name(f".{pptx.name}.{uuid.uuid4().hex}.tmp")
slide_count = 0
try:
with zipfile.ZipFile(pptx, "r") as source, zipfile.ZipFile(temporary, "w") as target:
target.comment = source.comment
for info in source.infolist():
data = source.read(info.filename)
if re.fullmatch(r"ppt/slides/slide\d+\.xml", info.filename):
data = replace_transition(data, transition)
slide_count += 1
target.writestr(info, data, compress_type=info.compress_type)
if slide_count == 0:
raise ExportError("exported PPTX contains no slide XML")
temporary.replace(pptx)
finally:
temporary.unlink(missing_ok=True)
return slide_count
def verify_output(pptx: Path, transition: str, expect_fonts: bool) -> Dict[str, Any]:
if not is_pptx(pptx):
raise ExportError(f"output is not a valid PPTX ZIP: {pptx}")
with zipfile.ZipFile(pptx) as archive:
broken = archive.testzip()
if broken:
raise ExportError(f"PPTX CRC check failed at: {broken}")
slide_names = [
name
for name in archive.namelist()
if re.fullmatch(r"ppt/slides/slide\d+\.xml", name)
]
slide_xml = {name: archive.read(name) for name in slide_names}
for data in slide_xml.values():
validate_transition_order(data, transition)
transition_hits = sum(has_direct_fade_transition(data) for data in slide_xml.values())
if transition == "fade" and transition_hits != len(slide_names):
raise ExportError("fade transition was not written to every slide")
fonts = [
name
for name in archive.namelist()
if name.startswith("ppt/fonts/") and not name.endswith("/")
]
if expect_fonts and not fonts:
log(
"warning: embed-fonts was enabled, but the official writer produced no font part"
)
return {
"slides": len(slide_names),
"fadeTransitions": transition_hits,
"fontParts": len(fonts),
"bytes": pptx.stat().st_size,
}
def serve(directory: Path) -> Tuple[ThreadingHTTPServer, threading.Thread, str]:
handler = lambda *args, **kwargs: QuietHandler( # noqa: E731
*args, directory=str(directory), **kwargs
)
server = ThreadingHTTPServer(("127.0.0.1", 0), handler)
thread = threading.Thread(target=server.serve_forever, daemon=True)
thread.start()
host, port = server.server_address
return server, thread, f"http://{host}:{port}/export_host.html"
def export_pptx(
source: Path,
output: Path,
transition: str,
embed_fonts: bool,
keep_download: bool = False,
force: bool = False,
) -> Dict[str, Any]:
manifest = find_manifest(source)
payload = build_payload(manifest)
output = output.expanduser().resolve()
output.parent.mkdir(parents=True, exist_ok=True)
if output.exists() and not force:
raise ExportError(f"output already exists (pass --force to replace it): {output}")
agent_browser = ensure_agent_browser()
log(f"manifest: {manifest}")
log(
f"defaults: transition={transition}, embed_fonts={'on' if embed_fonts else 'off'}"
)
with tempfile.TemporaryDirectory(prefix="open-kimi-ppt-export-") as temp_name:
temp_dir = Path(temp_name)
download_dir = temp_dir / "downloads"
download_dir.mkdir()
shutil.copy2(HOST_TEMPLATE, temp_dir / HOST_TEMPLATE.name)
(temp_dir / "payload.json").write_text(
json.dumps(payload, ensure_ascii=False), encoding="utf-8"
)
server, thread, url = serve(temp_dir)
session = f"open-kimi-ppt-export-{os.getpid()}-{uuid.uuid4().hex[:8]}"
browser = BrowserSession(agent_browser, session, temp_dir, download_dir)
try:
log("opening the public Kimi slide editor")
browser.open(url)
browser.run(
[
"wait",
"--fn",
'document.documentElement.dataset.deckStatus === "ready"',
],
timeout=120,
)
browser.run(["set", "viewport", "1280", "720"])
snapshot = browser.snapshot()
export_ref = ref_by_name(snapshot, "导出", "button")
browser.run(["click", f"@{export_ref}"])
dialog = wait_for_export_dialog(browser)
state = switch_state(dialog)
if state is not None:
switch_ref, checked, disabled = state
if disabled and checked != embed_fonts:
log("warning: the official font switch is disabled for this deck")
elif checked != embed_fonts:
browser.run(["click", f"@{switch_ref}"])
dialog = wait_for_export_dialog(browser)
elif embed_fonts:
log("warning: the official export dialog exposed no font switch")
download_ref = ref_by_name(dialog, "下载", "button")
log("generating PPTX in the browser")
result = browser.run(
["download", f"@{download_ref}", str(temp_dir / "browser-output.pptx")],
timeout=180,
check=False,
)
if result.returncode != 0:
log("download capture reported a timeout; checking browser output files")
downloaded = find_download((download_dir, temp_dir), timeout=90)
shutil.copy2(downloaded, output)
if keep_download:
debug_copy = output.with_name(f"{output.stem}.browser-raw.pptx")
if debug_copy.exists() and not force:
raise ExportError(
f"raw debug output already exists (pass --force): {debug_copy}"
)
shutil.copy2(downloaded, debug_copy)
finally:
browser.close()
server.shutdown()
server.server_close()
thread.join(timeout=2)
slide_count = patch_transitions(output, transition)
summary = verify_output(output, transition, embed_fonts)
summary["transitionPatchedSlides"] = slide_count
summary["output"] = str(output)
return summary
def parse_args(argv: Optional[Sequence[str]] = None) -> argparse.Namespace:
parser = argparse.ArgumentParser(
description=(
"Export a PPTD project to PPTX using Kimi's public browser-side writer. "
"Defaults: fade transition and embedded fonts."
)
)
parser.add_argument("input", type=Path, help=".pptd manifest or project directory")
parser.add_argument("--output", "-o", type=Path, help="output .pptx path")
parser.add_argument(
"--transition",
choices=("fade", "none"),
default="fade",
help="slide transition written to every slide (default: fade)",
)
font_group = parser.add_mutually_exclusive_group()
font_group.add_argument(
"--embed-fonts",
dest="embed_fonts",
action="store_true",
default=True,
help="embed fonts when available (default)",
)
font_group.add_argument(
"--no-embed-fonts",
dest="embed_fonts",
action="store_false",
help="disable font embedding",
)
parser.add_argument(
"--keep-browser-raw",
action="store_true",
help="also keep the unpatched browser download beside the output",
)
parser.add_argument(
"--force",
action="store_true",
help="replace an existing output file",
)
return parser.parse_args(argv)
def main(argv: Optional[Sequence[str]] = None) -> int:
args = parse_args(argv)
try:
manifest = find_manifest(args.input)
output = args.output or manifest.with_suffix(".pptx")
summary = export_pptx(
args.input,
output,
args.transition,
args.embed_fonts,
args.keep_browser_raw,
args.force,
)
except (ExportError, OSError, subprocess.SubprocessError) as exc:
print(f"open-kimi-ppt export failed: {exc}", file=sys.stderr)
return 1
print(json.dumps(summary, ensure_ascii=False, indent=2))
return 0
if __name__ == "__main__":
raise SystemExit(main())