Files
anonymous_github/src/core/AnonymizedFile.ts
T

516 lines
17 KiB
TypeScript

import { join, basename, dirname } from "path";
import { Response } from "express";
import { Readable } from "stream";
import { lookup } from "mime-types";
import got from "got";
import Repository from "./Repository";
import { RepositoryStatus } from "./types";
import config from "../config";
import {
anonymizePath,
hasCustomTermReplacement,
isTextFile,
} from "./anonymize-utils";
import AnonymousError from "./AnonymousError";
import { handleError } from "../server/routes/route-utils";
import FileModel from "./model/files/files.model";
import { IFile } from "./model/files/files.types";
import { FilterQuery } from "mongoose";
import { createLogger, serializeError } from "./logger";
import { githubTokenForStreamer } from "./github-token-context";
const logger = createLogger("anonymized-file");
function escapeRegex(value: string): string {
return value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
}
function defaultMaskCandidateRegex(value: string): RegExp {
const mask = new RegExp(
`${escapeRegex(config.ANONYMIZATION_MASK)}(?:-[0-9]+)?`,
"g"
);
let source = "^";
let lastIndex = 0;
let match: RegExpExecArray | null;
while ((match = mask.exec(value)) !== null) {
source += escapeRegex(value.slice(lastIndex, match.index));
source += "[^/]+";
lastIndex = match.index + match[0].length;
}
source += escapeRegex(value.slice(lastIndex)) + "$";
return new RegExp(source);
}
// Map a streamer error response to an AnonymousError that preserves the
// upstream status and error code instead of collapsing every failure into a
// generic 404. Without this, a corrupt cache, a 5xx from the streamer, an
// LFS pointer issue, and a missing file all surface to the user as the
// same `file_not_found` — which makes incidents impossible to triage.
function streamerErrorToAnonymous(
err: Error & { response?: { statusCode?: number; body?: unknown } },
context: { repoId: string; filePath: string }
): { error: AnonymousError; upstreamStatus?: number; upstreamBody?: string } {
const upstreamStatus = err?.response?.statusCode;
let errCode = "file_not_found";
let httpStatus = 404;
let upstreamBody: string | undefined;
if (err?.response?.body != null) {
try {
upstreamBody =
typeof err.response.body === "string"
? err.response.body
: Buffer.isBuffer(err.response.body)
? err.response.body.toString("utf8")
: JSON.stringify(err.response.body);
} catch {
// ignore body decode failures
}
if (upstreamBody) {
try {
const parsed = JSON.parse(upstreamBody);
if (parsed && typeof parsed.error === "string") {
errCode = parsed.error;
}
} catch {
// body wasn't JSON — keep the default code
}
}
}
if (typeof upstreamStatus === "number") {
// Pass through 4xx (client-meaningful: 404 file_not_found, 413
// file_too_big, 403 file_not_accessible). Collapse 5xx into 502 so
// browsers don't cache an upstream-fault as a missing-file 404.
httpStatus = upstreamStatus >= 500 ? 502 : upstreamStatus;
if (upstreamStatus >= 500 && errCode === "file_not_found") {
errCode = "streamer_upstream_error";
}
} else if (errCode === "file_not_found") {
// No HTTP response at all (connection refused, timeout, DNS) — that's
// a streamer fault, not a missing file.
errCode = "streamer_unreachable";
httpStatus = 502;
}
logger.warn("streamer fetch failed", {
code: errCode,
httpStatus,
repoId: context.repoId,
filePath: context.filePath,
upstreamStatus,
upstreamBody: upstreamBody?.slice(0, 500),
url: config.STREAMER_ENTRYPOINT
? join(config.STREAMER_ENTRYPOINT, "api")
: undefined,
err: serializeError(err),
});
return {
error: new AnonymousError(errCode, {
httpStatus,
cause: err,
}),
upstreamStatus,
upstreamBody,
};
}
/**
* Represent a file in a anonymized repository
*/
export default class AnonymizedFile {
repository: Repository;
anonymizedPath: string;
private _file?: IFile | null;
constructor(data: { repository: Repository; anonymizedPath: string }) {
this.repository = data.repository;
if (!this.repository.options.terms)
throw new AnonymousError("terms_not_specified", {
object: this,
httpStatus: 400,
});
this.anonymizedPath = data.anonymizedPath;
}
async sha() {
if (this._file) return this._file.sha?.replace(/"/g, "");
this._file = await this.getFileInfo();
return this._file.sha?.replace(/"/g, "");
}
async size(): Promise<number | undefined> {
if (this._file) return this._file.size;
this._file = await this.getFileInfo();
return this._file.size;
}
async getFileInfo(): Promise<IFile> {
if (this._file) return this._file;
let fileDir = dirname(this.anonymizedPath);
if (fileDir == ".") fileDir = "";
if (fileDir.endsWith("/")) fileDir = fileDir.slice(0, -1);
const filename = basename(this.anonymizedPath);
if (this.anonymizedPath == "") {
return {
name: "",
path: "",
repoId: this.repository.repoId,
};
}
// Always try the path verbatim first. Most paths contain no configured
// term, even when the repository uses custom replacements.
const exactQuery: FilterQuery<IFile> = {
repoId: this.repository.repoId,
path: fileDir,
};
if (filename != "") exactQuery.name = filename;
const exact = await FileModel.findOne(exactQuery);
if (exact) {
this._file = exact;
return exact;
}
const terms = this.repository.options.terms || [];
const usesDefaultMask = this.anonymizedPath.includes(
config.ANONYMIZATION_MASK
);
const usesCustomReplacement = hasCustomTermReplacement(terms);
if (!usesDefaultMask && !usesCustomReplacement) {
// The stored tree can be incomplete: GitHub truncates tree listings of
// very large repositories, and folders recorded in `truncatedFolders`
// have entries that never made it into the database. Ask GitHub
// directly for the path before concluding the file does not exist
// (#738). Without an anonymization mask the anonymized path is the
// original path, so it can be looked up as-is.
const recovered = await this.recoverTruncatedFile(fileDir);
if (recovered) {
this._file = recovered;
return recovered;
}
throw new AnonymousError("file_not_found", {
object: this,
httpStatus: 404,
});
}
// Custom replacements do not carry a marker that can be reversed into a
// narrow Mongo query. Fetch the repository's paths and verify them by
// re-applying anonymization. Default XXXX-N masks retain the optimized,
// anchored query.
const candidates = usesCustomReplacement
? await FileModel.find({ repoId: this.repository.repoId }).exec()
: await FileModel.find({
repoId: this.repository.repoId,
path: defaultMaskCandidateRegex(fileDir),
name: defaultMaskCandidateRegex(filename),
}).exec();
for (const candidate of candidates) {
const candidatePath = join(candidate.path, candidate.name);
if (
anonymizePath(candidatePath, terms) == this.anonymizedPath
) {
this._file = candidate;
return candidate;
}
}
// If applying the configured terms does not alter the requested path, it
// may simply be absent from a truncated tree and can be recovered as-is.
if (anonymizePath(this.anonymizedPath, terms) === this.anonymizedPath) {
const recovered = await this.recoverTruncatedFile(fileDir);
if (recovered) {
this._file = recovered;
return recovered;
}
}
throw new AnonymousError("file_not_found", {
object: this,
httpStatus: 404,
});
}
// On-demand recovery for files missing from the database because the
// GitHub tree listing was truncated. Only paths under a recorded
// truncated folder qualify — everything else is a genuine miss.
private async recoverTruncatedFile(fileDir: string): Promise<IFile | null> {
const truncated = this.repository.model.truncatedFolders || [];
const isAffected = truncated.some(
(folder) =>
folder === "" || fileDir === folder || fileDir.startsWith(folder + "/")
);
if (!isAffected) return null;
const source = this.repository.source as {
fetchFileInfoFromPath?: (filePath: string) => Promise<IFile | null>;
};
if (typeof source.fetchFileInfoFromPath !== "function") return null;
const recovered = await source.fetchFileInfoFromPath(this.anonymizedPath);
if (!recovered) return null;
recovered.repoId = this.repository.repoId;
logger.info("recovered file from truncated tree", {
repoId: this.repository.repoId,
path: this.anonymizedPath,
});
try {
// Cache it so the next request is served from the database.
await FileModel.create(recovered);
} catch (error) {
logger.warn(
"failed to cache recovered file",
serializeError(error as Error)
);
}
return recovered;
}
/**
* De-anonymize the path
*
* @returns the origin relative path of the file
*/
async originalPath(): Promise<string> {
if (this.anonymizedPath == null) {
throw new AnonymousError("path_not_specified", {
object: this,
httpStatus: 400,
});
}
if (!this._file) {
this._file = await this.getFileInfo();
}
return join(this._file.path, this._file.name);
}
extension() {
const filename = basename(this._file?.name || this.anonymizedPath);
const extensions = filename.split(".").reverse();
return extensions[0].toLowerCase();
}
isImage() {
const extension = this.extension();
return [
"png",
"jpg",
"jpeg",
"gif",
"svg",
"ico",
"bmp",
"tiff",
"tif",
"webp",
"avif",
"heif",
"heic",
].includes(extension);
}
isFileSupported() {
const extension = this.extension();
if (!this.repository.options.pdf && extension == "pdf") {
return false;
}
if (!this.repository.options.image && this.isImage()) {
return false;
}
return true;
}
async content(): Promise<Readable> {
if (this.anonymizedPath.includes(config.ANONYMIZATION_MASK)) {
await this.originalPath();
}
if (this._file?.size && this._file?.size > config.MAX_FILE_SIZE) {
throw new AnonymousError("file_too_big", {
object: this,
httpStatus: 413,
});
}
const content = await this.repository.source?.getFileContent(this);
const cacheWasReset = this.repository.model.isReseted;
if (cacheWasReset) {
await this.repository.markCachePresent();
}
if (cacheWasReset || this.repository.status != RepositoryStatus.READY) {
await this.repository.updateStatus(RepositoryStatus.READY);
}
return content;
}
async anonymizedContent() {
const anonymizer = this.repository.generateAnonymizeTransformer(
await this.originalPath()
);
if (!config.STREAMER_ENTRYPOINT) {
// collect the content locally
const content = await this.content();
content.on("error", (err) => anonymizer.destroy(err));
return content.pipe(anonymizer);
}
// use the streamer service
return got.stream(join(config.STREAMER_ENTRYPOINT, "api"), {
method: "POST",
json: {
token: await githubTokenForStreamer(await this.repository.getToken(), this.repository.model.source.repositoryName),
repoFullName: this.repository.model.source.repositoryName,
commit: this.repository.model.source.commit,
branch: this.repository.model.source.branch,
repoId: this.repository.repoId,
filePath: this.filePath,
sha: await this.sha(),
size: await this.size(),
anonymizerOptions: anonymizer.opt,
},
});
}
get filePath() {
if (!this._file) {
if (this.anonymizedPath.includes(config.ANONYMIZATION_MASK)) {
throw new AnonymousError("path_not_defined", {
object: this,
httpStatus: 400,
});
}
return this.anonymizedPath;
}
return join(this._file.path, this._file.name);
}
async send(res: Response): Promise<void> {
const anonymizer = this.repository.generateAnonymizeTransformer(
await this.originalPath()
);
// eslint-disable-next-line no-async-promise-executor
return new Promise<void>(async (resolve, reject) => {
try {
if (config.STREAMER_ENTRYPOINT) {
// use the streamer service
const [sha, size, token] = await Promise.all([
this.sha(),
this.size(),
this.repository.getToken(),
]);
const resStream = got
.stream(join(config.STREAMER_ENTRYPOINT, "api"), {
method: "POST",
json: {
sha,
size,
token: await githubTokenForStreamer(token, this.repository.model.source.repositoryName),
repoFullName: this.repository.model.source.repositoryName,
commit: this.repository.model.source.commit,
branch: this.repository.model.source.branch,
repoId: this.repository.repoId,
filePath: this.filePath,
anonymizerOptions: anonymizer.opt,
},
})
.on("error", (err: Error) => {
const { error } = streamerErrorToAnonymous(
err as Error & {
response?: { statusCode?: number; body?: unknown };
},
{
repoId: this.repository.repoId,
filePath: this.anonymizedPath,
}
);
error.value = this;
handleError(error, res);
});
// Forward Content-Type from the streamer's upstream response.
// got.stream(...).pipe(res) forwards body bytes only — without
// this, the parent response has no Content-Type and the browser
// guesses (text renders as download, images as octet-stream).
resStream.on("response", (upstream: { headers: Record<string, string | string[] | undefined> }) => {
if (res.headersSent) return;
const ct = upstream.headers["content-type"];
if (typeof ct === "string") {
res.contentType(ct);
} else {
const fallback = lookup(this.anonymizedPath);
if (fallback) res.contentType(fallback);
else if (isTextFile(this.anonymizedPath)) res.contentType("text/plain");
}
});
resStream.pipe(res);
// Resolve as soon as the response is fully written rather than
// waiting for the socket to close — keep-alive sockets stay open
// long after the body is delivered, and we don't want to delay
// post-send work like countView() that long.
res.on("finish", () => {
resolve();
});
res.on("close", () => {
resolve();
});
res.on("error", (err) => {
reject(err);
});
return;
}
const mime = lookup(this.anonymizedPath);
if (mime && this.extension() != "ts") {
res.contentType(mime);
} else if (isTextFile(this.anonymizedPath)) {
res.contentType("text/plain");
}
// For text files we anonymize on the fly and the output length can
// differ from the upstream, so byte ranges aren't meaningful — keep
// Accept-Ranges: none. For binary files (images, video, archives)
// the transformer is a passthrough, so omitting the explicit "none"
// lets <video>/<audio> elements use the standard fallback to a full
// download instead of refusing to play (#538).
const isTextEntry = isTextFile(this.anonymizedPath) === true;
if (isTextEntry) {
res.header("Accept-Ranges", "none");
}
anonymizer.once("transform", (data) => {
if (!mime && data.isText) {
res.contentType("text/plain");
}
});
const content = await this.content();
function handleStreamError(error: Error) {
if (!content.closed && !content.destroyed) {
content.destroy();
}
reject(error);
}
content
.on("error", handleStreamError)
.pipe(anonymizer)
.on("error", handleStreamError)
.pipe(res)
.on("error", handleStreamError)
.on("finish", () => {
// resolve on body fully written rather than waiting for the
// socket to close — keep-alive can hold the socket open long
// after the response is delivered, delaying post-send work.
if (!content.closed && !content.destroyed) {
content.destroy();
}
resolve();
})
.on("close", () => {
if (!content.closed && !content.destroyed) {
content.destroy();
}
resolve();
});
} catch (error) {
reject(error);
}
});
}
}