diff --git a/src/app/api/workspace/download/route.ts b/src/app/api/workspace/download/route.ts new file mode 100644 index 0000000..37d8409 --- /dev/null +++ b/src/app/api/workspace/download/route.ts @@ -0,0 +1,112 @@ +import { createReadStream, promises as fs } from "fs"; +import { tmpdir } from "os"; +import { join } from "path"; +import { randomUUID } from "crypto"; +import { spawn } from "child_process"; +import { Readable } from "stream"; +import { NextRequest, NextResponse } from "next/server"; +import { + getWorkspaceDir, + zipExcludeArgs, + isCrossOrigin, +} from "@/lib/server/workspace"; + +export const runtime = "nodejs"; + +/** Zip `workspaceDir` (minus the ignore list) into `outFile` using the OS `zip`. + * Aborts (and kills the child) if `signal` fires — e.g. the client disconnects + * mid-archive. */ +function zipWorkspace( + workspaceDir: string, + outFile: string, + signal?: AbortSignal +): Promise { + return new Promise((resolve, reject) => { + // -r recurse, -q quiet, -X drop extra file attributes, -y store symlinks AS + // symlinks instead of dereferencing them (so a symlink pointing outside the + // workspace can't pull external file *contents* into the archive). + // Exclusions come from the shared ignore lists so the archive matches the + // tree exactly (dotfiles, large_tool_results/conversation_history, build noise). + const child = spawn( + "zip", + ["-r", "-q", "-X", "-y", outFile, ".", ...zipExcludeArgs()], + { cwd: workspaceDir } + ); + + const onAbort = () => child.kill("SIGKILL"); + if (signal) { + if (signal.aborted) { + child.kill("SIGKILL"); + reject(new Error("Request aborted.")); + return; + } + signal.addEventListener("abort", onAbort, { once: true }); + } + + let stderr = ""; + child.stderr.on("data", (d) => (stderr += d.toString())); + child.on("error", (err) => { + signal?.removeEventListener("abort", onAbort); + reject( + (err as NodeJS.ErrnoException).code === "ENOENT" + ? new Error( + "The `zip` command is not available on this system, so the workspace can't be downloaded as an archive." + ) + : err + ); + }); + child.on("close", (code) => { + signal?.removeEventListener("abort", onAbort); + if (signal?.aborted) reject(new Error("Request aborted.")); + // 12 = "nothing to do" (empty workspace) — treat as a friendly error. + else if (code === 12) reject(new Error("The workspace is empty.")); + else if (code !== 0) + reject(new Error(stderr.trim() || `zip exited with code ${code}`)); + else resolve(); + }); + }); +} + +export async function GET(request: NextRequest) { + let tmpFile: string | null = null; + try { + if (isCrossOrigin(request)) { + return NextResponse.json( + { error: "Cross-origin workspace access is not allowed." }, + { status: 403 } + ); + } + + const workspaceDir = await getWorkspaceDir(); + tmpFile = join(tmpdir(), `evoscientist-workspace-${randomUUID()}.zip`); + await zipWorkspace(workspaceDir, tmpFile, request.signal); + + const stat = await fs.stat(tmpFile); + const nodeStream = createReadStream(tmpFile); + // Delete the temp archive once the response has been fully read (or the + // client disconnects) — `close` fires in both cases. + const cleanup = tmpFile; + nodeStream.on("close", () => void fs.rm(cleanup, { force: true })); + const webStream = Readable.toWeb(nodeStream) as ReadableStream; + + return new NextResponse(webStream, { + headers: { + "Content-Type": "application/zip", + "Content-Length": String(stat.size), + "Content-Disposition": 'attachment; filename="workspace.zip"', + "Cache-Control": "no-store", + }, + }); + } catch (error) { + if (tmpFile) await fs.rm(tmpFile, { force: true }).catch(() => {}); + return NextResponse.json( + { + error: + error instanceof Error + ? error.message + : "Failed to package the workspace.", + }, + { status: 400 } + ); + } +} diff --git a/src/app/api/workspace/file/route.ts b/src/app/api/workspace/file/route.ts new file mode 100644 index 0000000..4f412cb --- /dev/null +++ b/src/app/api/workspace/file/route.ts @@ -0,0 +1,120 @@ +import { createReadStream } from "fs"; +import { promises as fs } from "fs"; +import { basename, extname } from "path"; +import { Readable } from "stream"; +import { NextRequest, NextResponse } from "next/server"; +import { + getWorkspaceDir, + safeResolve, + isCrossOrigin, +} from "@/lib/server/workspace"; + +/** RFC 6266 Content-Disposition value with both an ASCII fallback and a UTF-8 + * `filename*` so non-ASCII names (e.g. Chinese) download with their real name + * instead of percent-encoded gibberish. */ +function contentDisposition(fileName: string, asAttachment: boolean): string { + const ascii = fileName.replace(/[^\x20-\x7e]/g, "_").replace(/["\\]/g, "_"); + const encoded = encodeURIComponent(fileName).replace( + /['()*]/g, + (c) => "%" + c.charCodeAt(0).toString(16).toUpperCase() + ); + return `${asAttachment ? "attachment" : "inline"}; filename="${ascii}"; filename*=UTF-8''${encoded}`; +} + +export const runtime = "nodejs"; + +// Map common research-output extensions to a Content-Type. Anything unlisted is +// served as a download (octet-stream) so the browser never tries to execute it. +const CONTENT_TYPES: Record = { + txt: "text/plain; charset=utf-8", + md: "text/markdown; charset=utf-8", + log: "text/plain; charset=utf-8", + csv: "text/csv; charset=utf-8", + tsv: "text/tab-separated-values; charset=utf-8", + json: "application/json; charset=utf-8", + py: "text/plain; charset=utf-8", + js: "text/plain; charset=utf-8", + ts: "text/plain; charset=utf-8", + tsx: "text/plain; charset=utf-8", + sh: "text/plain; charset=utf-8", + yaml: "text/plain; charset=utf-8", + yml: "text/plain; charset=utf-8", + toml: "text/plain; charset=utf-8", + tex: "text/plain; charset=utf-8", + bib: "text/plain; charset=utf-8", + html: "text/plain; charset=utf-8", // never text/html — don't let it render + css: "text/plain; charset=utf-8", + xml: "text/plain; charset=utf-8", + png: "image/png", + jpg: "image/jpeg", + jpeg: "image/jpeg", + gif: "image/gif", + webp: "image/webp", + svg: "image/svg+xml", + bmp: "image/bmp", + pdf: "application/pdf", +}; + +export async function GET(request: NextRequest) { + try { + if (isCrossOrigin(request)) { + return NextResponse.json( + { error: "Cross-origin workspace access is not allowed." }, + { status: 403 } + ); + } + + const relPath = request.nextUrl.searchParams.get("path"); + if (!relPath) { + return NextResponse.json({ error: "Missing path." }, { status: 400 }); + } + const download = request.nextUrl.searchParams.get("download") === "1"; + + const workspaceDir = await getWorkspaceDir(); + // safeResolve canonicalizes + re-checks containment, so a symlink can't be + // used to read a file outside the workspace (or a hidden/internal entry). + const target = await safeResolve(workspaceDir, relPath); + + const stat = await fs.stat(target); + if (!stat.isFile()) { + return NextResponse.json( + { error: "Not a file." }, + { status: 400 } + ); + } + + const ext = extname(target).slice(1).toLowerCase(); + const contentType = CONTENT_TYPES[ext] ?? "application/octet-stream"; + // Octet-stream and explicit ?download=1 go out as attachments; previewable + // types render inline. + const asAttachment = download || contentType === "application/octet-stream"; + + const nodeStream = createReadStream(target); + const webStream = Readable.toWeb(nodeStream) as ReadableStream; + + const fileName = basename(target); + return new NextResponse(webStream, { + headers: { + "Content-Type": contentType, + "Content-Length": String(stat.size), + "Content-Disposition": contentDisposition(fileName, asAttachment), + // Workspace files are agent/user-controlled. `sandbox` neutralizes + // scripts in an inline SVG/HTML opened directly (XSS), and `nosniff` + // stops the browser from sniffing a text/* file into executable HTML. + // Neither affects /