import { join, basename, dirname } from "path"; import { Response } from "express"; import { Readable } from "stream"; import { lookup } from "mime-types"; import got from "got"; import Repository from "./Repository"; import { RepositoryStatus } from "./types"; import config from "../config"; import { anonymizePath, hasCustomTermReplacement, isTextFile, } from "./anonymize-utils"; import AnonymousError from "./AnonymousError"; import { handleError } from "../server/routes/route-utils"; import FileModel from "./model/files/files.model"; import { IFile } from "./model/files/files.types"; import { FilterQuery } from "mongoose"; import { createLogger, serializeError } from "./logger"; import { githubTokenForStreamer } from "./github-token-context"; const logger = createLogger("anonymized-file"); function escapeRegex(value: string): string { return value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); } function defaultMaskCandidateRegex(value: string): RegExp { const mask = new RegExp( `${escapeRegex(config.ANONYMIZATION_MASK)}(?:-[0-9]+)?`, "g" ); let source = "^"; let lastIndex = 0; let match: RegExpExecArray | null; while ((match = mask.exec(value)) !== null) { source += escapeRegex(value.slice(lastIndex, match.index)); source += "[^/]+"; lastIndex = match.index + match[0].length; } source += escapeRegex(value.slice(lastIndex)) + "$"; return new RegExp(source); } // Map a streamer error response to an AnonymousError that preserves the // upstream status and error code instead of collapsing every failure into a // generic 404. Without this, a corrupt cache, a 5xx from the streamer, an // LFS pointer issue, and a missing file all surface to the user as the // same `file_not_found` — which makes incidents impossible to triage. function streamerErrorToAnonymous( err: Error & { response?: { statusCode?: number; body?: unknown } }, context: { repoId: string; filePath: string } ): { error: AnonymousError; upstreamStatus?: number; upstreamBody?: string } { const upstreamStatus = err?.response?.statusCode; let errCode = "file_not_found"; let httpStatus = 404; let upstreamBody: string | undefined; if (err?.response?.body != null) { try { upstreamBody = typeof err.response.body === "string" ? err.response.body : Buffer.isBuffer(err.response.body) ? err.response.body.toString("utf8") : JSON.stringify(err.response.body); } catch { // ignore body decode failures } if (upstreamBody) { try { const parsed = JSON.parse(upstreamBody); if (parsed && typeof parsed.error === "string") { errCode = parsed.error; } } catch { // body wasn't JSON — keep the default code } } } if (typeof upstreamStatus === "number") { // Pass through 4xx (client-meaningful: 404 file_not_found, 413 // file_too_big, 403 file_not_accessible). Collapse 5xx into 502 so // browsers don't cache an upstream-fault as a missing-file 404. httpStatus = upstreamStatus >= 500 ? 502 : upstreamStatus; if (upstreamStatus >= 500 && errCode === "file_not_found") { errCode = "streamer_upstream_error"; } } else if (errCode === "file_not_found") { // No HTTP response at all (connection refused, timeout, DNS) — that's // a streamer fault, not a missing file. errCode = "streamer_unreachable"; httpStatus = 502; } logger.warn("streamer fetch failed", { code: errCode, httpStatus, repoId: context.repoId, filePath: context.filePath, upstreamStatus, upstreamBody: upstreamBody?.slice(0, 500), url: config.STREAMER_ENTRYPOINT ? join(config.STREAMER_ENTRYPOINT, "api") : undefined, err: serializeError(err), }); return { error: new AnonymousError(errCode, { httpStatus, cause: err, }), upstreamStatus, upstreamBody, }; } /** * Represent a file in a anonymized repository */ export default class AnonymizedFile { repository: Repository; anonymizedPath: string; private _file?: IFile | null; constructor(data: { repository: Repository; anonymizedPath: string }) { this.repository = data.repository; if (!this.repository.options.terms) throw new AnonymousError("terms_not_specified", { object: this, httpStatus: 400, }); this.anonymizedPath = data.anonymizedPath; } async sha() { if (this._file) return this._file.sha?.replace(/"/g, ""); this._file = await this.getFileInfo(); return this._file.sha?.replace(/"/g, ""); } async size(): Promise { if (this._file) return this._file.size; this._file = await this.getFileInfo(); return this._file.size; } async getFileInfo(): Promise { if (this._file) return this._file; let fileDir = dirname(this.anonymizedPath); if (fileDir == ".") fileDir = ""; if (fileDir.endsWith("/")) fileDir = fileDir.slice(0, -1); const filename = basename(this.anonymizedPath); if (this.anonymizedPath == "") { return { name: "", path: "", repoId: this.repository.repoId, }; } // Always try the path verbatim first. Most paths contain no configured // term, even when the repository uses custom replacements. const exactQuery: FilterQuery = { repoId: this.repository.repoId, path: fileDir, }; if (filename != "") exactQuery.name = filename; const exact = await FileModel.findOne(exactQuery); if (exact) { this._file = exact; return exact; } const terms = this.repository.options.terms || []; const usesDefaultMask = this.anonymizedPath.includes( config.ANONYMIZATION_MASK ); const usesCustomReplacement = hasCustomTermReplacement(terms); if (!usesDefaultMask && !usesCustomReplacement) { // The stored tree can be incomplete: GitHub truncates tree listings of // very large repositories, and folders recorded in `truncatedFolders` // have entries that never made it into the database. Ask GitHub // directly for the path before concluding the file does not exist // (#738). Without an anonymization mask the anonymized path is the // original path, so it can be looked up as-is. const recovered = await this.recoverTruncatedFile(fileDir); if (recovered) { this._file = recovered; return recovered; } throw new AnonymousError("file_not_found", { object: this, httpStatus: 404, }); } // Custom replacements do not carry a marker that can be reversed into a // narrow Mongo query. Fetch the repository's paths and verify them by // re-applying anonymization. Default XXXX-N masks retain the optimized, // anchored query. const candidates = usesCustomReplacement ? await FileModel.find({ repoId: this.repository.repoId }).exec() : await FileModel.find({ repoId: this.repository.repoId, path: defaultMaskCandidateRegex(fileDir), name: defaultMaskCandidateRegex(filename), }).exec(); for (const candidate of candidates) { const candidatePath = join(candidate.path, candidate.name); if ( anonymizePath(candidatePath, terms) == this.anonymizedPath ) { this._file = candidate; return candidate; } } // If applying the configured terms does not alter the requested path, it // may simply be absent from a truncated tree and can be recovered as-is. if (anonymizePath(this.anonymizedPath, terms) === this.anonymizedPath) { const recovered = await this.recoverTruncatedFile(fileDir); if (recovered) { this._file = recovered; return recovered; } } throw new AnonymousError("file_not_found", { object: this, httpStatus: 404, }); } // On-demand recovery for files missing from the database because the // GitHub tree listing was truncated. Only paths under a recorded // truncated folder qualify — everything else is a genuine miss. private async recoverTruncatedFile(fileDir: string): Promise { const truncated = this.repository.model.truncatedFolders || []; const isAffected = truncated.some( (folder) => folder === "" || fileDir === folder || fileDir.startsWith(folder + "/") ); if (!isAffected) return null; const source = this.repository.source as { fetchFileInfoFromPath?: (filePath: string) => Promise; }; if (typeof source.fetchFileInfoFromPath !== "function") return null; const recovered = await source.fetchFileInfoFromPath(this.anonymizedPath); if (!recovered) return null; recovered.repoId = this.repository.repoId; logger.info("recovered file from truncated tree", { repoId: this.repository.repoId, path: this.anonymizedPath, }); try { // Cache it so the next request is served from the database. await FileModel.create(recovered); } catch (error) { logger.warn( "failed to cache recovered file", serializeError(error as Error) ); } return recovered; } /** * De-anonymize the path * * @returns the origin relative path of the file */ async originalPath(): Promise { if (this.anonymizedPath == null) { throw new AnonymousError("path_not_specified", { object: this, httpStatus: 400, }); } if (!this._file) { this._file = await this.getFileInfo(); } return join(this._file.path, this._file.name); } extension() { const filename = basename(this._file?.name || this.anonymizedPath); const extensions = filename.split(".").reverse(); return extensions[0].toLowerCase(); } isImage() { const extension = this.extension(); return [ "png", "jpg", "jpeg", "gif", "svg", "ico", "bmp", "tiff", "tif", "webp", "avif", "heif", "heic", ].includes(extension); } isFileSupported() { const extension = this.extension(); if (!this.repository.options.pdf && extension == "pdf") { return false; } if (!this.repository.options.image && this.isImage()) { return false; } return true; } async content(): Promise { if (this.anonymizedPath.includes(config.ANONYMIZATION_MASK)) { await this.originalPath(); } if (this._file?.size && this._file?.size > config.MAX_FILE_SIZE) { throw new AnonymousError("file_too_big", { object: this, httpStatus: 413, }); } const content = await this.repository.source?.getFileContent(this); const cacheWasReset = this.repository.model.isReseted; if (cacheWasReset) { await this.repository.markCachePresent(); } if (cacheWasReset || this.repository.status != RepositoryStatus.READY) { await this.repository.updateStatus(RepositoryStatus.READY); } return content; } async anonymizedContent() { const anonymizer = this.repository.generateAnonymizeTransformer( await this.originalPath() ); if (!config.STREAMER_ENTRYPOINT) { // collect the content locally const content = await this.content(); content.on("error", (err) => anonymizer.destroy(err)); return content.pipe(anonymizer); } // use the streamer service return got.stream(join(config.STREAMER_ENTRYPOINT, "api"), { method: "POST", json: { token: await githubTokenForStreamer(await this.repository.getToken(), this.repository.model.source.repositoryName), repoFullName: this.repository.model.source.repositoryName, commit: this.repository.model.source.commit, branch: this.repository.model.source.branch, repoId: this.repository.repoId, filePath: this.filePath, sha: await this.sha(), size: await this.size(), anonymizerOptions: anonymizer.opt, }, }); } get filePath() { if (!this._file) { if (this.anonymizedPath.includes(config.ANONYMIZATION_MASK)) { throw new AnonymousError("path_not_defined", { object: this, httpStatus: 400, }); } return this.anonymizedPath; } return join(this._file.path, this._file.name); } async send(res: Response): Promise { const anonymizer = this.repository.generateAnonymizeTransformer( await this.originalPath() ); // eslint-disable-next-line no-async-promise-executor return new Promise(async (resolve, reject) => { try { if (config.STREAMER_ENTRYPOINT) { // use the streamer service const [sha, size, token] = await Promise.all([ this.sha(), this.size(), this.repository.getToken(), ]); const resStream = got .stream(join(config.STREAMER_ENTRYPOINT, "api"), { method: "POST", json: { sha, size, token: await githubTokenForStreamer(token, this.repository.model.source.repositoryName), repoFullName: this.repository.model.source.repositoryName, commit: this.repository.model.source.commit, branch: this.repository.model.source.branch, repoId: this.repository.repoId, filePath: this.filePath, anonymizerOptions: anonymizer.opt, }, }) .on("error", (err: Error) => { const { error } = streamerErrorToAnonymous( err as Error & { response?: { statusCode?: number; body?: unknown }; }, { repoId: this.repository.repoId, filePath: this.anonymizedPath, } ); error.value = this; handleError(error, res); }); // Forward Content-Type from the streamer's upstream response. // got.stream(...).pipe(res) forwards body bytes only — without // this, the parent response has no Content-Type and the browser // guesses (text renders as download, images as octet-stream). resStream.on("response", (upstream: { headers: Record }) => { if (res.headersSent) return; const ct = upstream.headers["content-type"]; if (typeof ct === "string") { res.contentType(ct); } else { const fallback = lookup(this.anonymizedPath); if (fallback) res.contentType(fallback); else if (isTextFile(this.anonymizedPath)) res.contentType("text/plain"); } }); resStream.pipe(res); // Resolve as soon as the response is fully written rather than // waiting for the socket to close — keep-alive sockets stay open // long after the body is delivered, and we don't want to delay // post-send work like countView() that long. res.on("finish", () => { resolve(); }); res.on("close", () => { resolve(); }); res.on("error", (err) => { reject(err); }); return; } const mime = lookup(this.anonymizedPath); if (mime && this.extension() != "ts") { res.contentType(mime); } else if (isTextFile(this.anonymizedPath)) { res.contentType("text/plain"); } // For text files we anonymize on the fly and the output length can // differ from the upstream, so byte ranges aren't meaningful — keep // Accept-Ranges: none. For binary files (images, video, archives) // the transformer is a passthrough, so omitting the explicit "none" // lets