mirror of
https://github.com/tdurieux/anonymous_github.git
synced 2026-09-29 13:41:45 +02:00
516 lines
17 KiB
TypeScript
516 lines
17 KiB
TypeScript
import { join, basename, dirname } from "path";
|
|
import { Response } from "express";
|
|
import { Readable } from "stream";
|
|
import { lookup } from "mime-types";
|
|
import got from "got";
|
|
|
|
import Repository from "./Repository";
|
|
import { RepositoryStatus } from "./types";
|
|
import config from "../config";
|
|
import {
|
|
anonymizePath,
|
|
hasCustomTermReplacement,
|
|
isTextFile,
|
|
} from "./anonymize-utils";
|
|
import AnonymousError from "./AnonymousError";
|
|
import { handleError } from "../server/routes/route-utils";
|
|
import FileModel from "./model/files/files.model";
|
|
import { IFile } from "./model/files/files.types";
|
|
import { FilterQuery } from "mongoose";
|
|
import { createLogger, serializeError } from "./logger";
|
|
import { githubTokenForStreamer } from "./github-token-context";
|
|
|
|
const logger = createLogger("anonymized-file");
|
|
|
|
function escapeRegex(value: string): string {
|
|
return value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
}
|
|
|
|
function defaultMaskCandidateRegex(value: string): RegExp {
|
|
const mask = new RegExp(
|
|
`${escapeRegex(config.ANONYMIZATION_MASK)}(?:-[0-9]+)?`,
|
|
"g"
|
|
);
|
|
let source = "^";
|
|
let lastIndex = 0;
|
|
let match: RegExpExecArray | null;
|
|
while ((match = mask.exec(value)) !== null) {
|
|
source += escapeRegex(value.slice(lastIndex, match.index));
|
|
source += "[^/]+";
|
|
lastIndex = match.index + match[0].length;
|
|
}
|
|
source += escapeRegex(value.slice(lastIndex)) + "$";
|
|
return new RegExp(source);
|
|
}
|
|
|
|
// Map a streamer error response to an AnonymousError that preserves the
|
|
// upstream status and error code instead of collapsing every failure into a
|
|
// generic 404. Without this, a corrupt cache, a 5xx from the streamer, an
|
|
// LFS pointer issue, and a missing file all surface to the user as the
|
|
// same `file_not_found` — which makes incidents impossible to triage.
|
|
function streamerErrorToAnonymous(
|
|
err: Error & { response?: { statusCode?: number; body?: unknown } },
|
|
context: { repoId: string; filePath: string }
|
|
): { error: AnonymousError; upstreamStatus?: number; upstreamBody?: string } {
|
|
const upstreamStatus = err?.response?.statusCode;
|
|
let errCode = "file_not_found";
|
|
let httpStatus = 404;
|
|
let upstreamBody: string | undefined;
|
|
|
|
if (err?.response?.body != null) {
|
|
try {
|
|
upstreamBody =
|
|
typeof err.response.body === "string"
|
|
? err.response.body
|
|
: Buffer.isBuffer(err.response.body)
|
|
? err.response.body.toString("utf8")
|
|
: JSON.stringify(err.response.body);
|
|
} catch {
|
|
// ignore body decode failures
|
|
}
|
|
if (upstreamBody) {
|
|
try {
|
|
const parsed = JSON.parse(upstreamBody);
|
|
if (parsed && typeof parsed.error === "string") {
|
|
errCode = parsed.error;
|
|
}
|
|
} catch {
|
|
// body wasn't JSON — keep the default code
|
|
}
|
|
}
|
|
}
|
|
if (typeof upstreamStatus === "number") {
|
|
// Pass through 4xx (client-meaningful: 404 file_not_found, 413
|
|
// file_too_big, 403 file_not_accessible). Collapse 5xx into 502 so
|
|
// browsers don't cache an upstream-fault as a missing-file 404.
|
|
httpStatus = upstreamStatus >= 500 ? 502 : upstreamStatus;
|
|
if (upstreamStatus >= 500 && errCode === "file_not_found") {
|
|
errCode = "streamer_upstream_error";
|
|
}
|
|
} else if (errCode === "file_not_found") {
|
|
// No HTTP response at all (connection refused, timeout, DNS) — that's
|
|
// a streamer fault, not a missing file.
|
|
errCode = "streamer_unreachable";
|
|
httpStatus = 502;
|
|
}
|
|
|
|
logger.warn("streamer fetch failed", {
|
|
code: errCode,
|
|
httpStatus,
|
|
repoId: context.repoId,
|
|
filePath: context.filePath,
|
|
upstreamStatus,
|
|
upstreamBody: upstreamBody?.slice(0, 500),
|
|
url: config.STREAMER_ENTRYPOINT
|
|
? join(config.STREAMER_ENTRYPOINT, "api")
|
|
: undefined,
|
|
err: serializeError(err),
|
|
});
|
|
|
|
return {
|
|
error: new AnonymousError(errCode, {
|
|
httpStatus,
|
|
cause: err,
|
|
}),
|
|
upstreamStatus,
|
|
upstreamBody,
|
|
};
|
|
}
|
|
|
|
/**
|
|
* Represent a file in a anonymized repository
|
|
*/
|
|
export default class AnonymizedFile {
|
|
repository: Repository;
|
|
anonymizedPath: string;
|
|
|
|
private _file?: IFile | null;
|
|
|
|
constructor(data: { repository: Repository; anonymizedPath: string }) {
|
|
this.repository = data.repository;
|
|
if (!this.repository.options.terms)
|
|
throw new AnonymousError("terms_not_specified", {
|
|
object: this,
|
|
httpStatus: 400,
|
|
});
|
|
this.anonymizedPath = data.anonymizedPath;
|
|
}
|
|
|
|
async sha() {
|
|
if (this._file) return this._file.sha?.replace(/"/g, "");
|
|
this._file = await this.getFileInfo();
|
|
return this._file.sha?.replace(/"/g, "");
|
|
}
|
|
|
|
async size(): Promise<number | undefined> {
|
|
if (this._file) return this._file.size;
|
|
this._file = await this.getFileInfo();
|
|
return this._file.size;
|
|
}
|
|
|
|
async getFileInfo(): Promise<IFile> {
|
|
if (this._file) return this._file;
|
|
let fileDir = dirname(this.anonymizedPath);
|
|
if (fileDir == ".") fileDir = "";
|
|
if (fileDir.endsWith("/")) fileDir = fileDir.slice(0, -1);
|
|
const filename = basename(this.anonymizedPath);
|
|
|
|
if (this.anonymizedPath == "") {
|
|
return {
|
|
name: "",
|
|
path: "",
|
|
repoId: this.repository.repoId,
|
|
};
|
|
}
|
|
|
|
// Always try the path verbatim first. Most paths contain no configured
|
|
// term, even when the repository uses custom replacements.
|
|
const exactQuery: FilterQuery<IFile> = {
|
|
repoId: this.repository.repoId,
|
|
path: fileDir,
|
|
};
|
|
if (filename != "") exactQuery.name = filename;
|
|
const exact = await FileModel.findOne(exactQuery);
|
|
if (exact) {
|
|
this._file = exact;
|
|
return exact;
|
|
}
|
|
|
|
const terms = this.repository.options.terms || [];
|
|
const usesDefaultMask = this.anonymizedPath.includes(
|
|
config.ANONYMIZATION_MASK
|
|
);
|
|
const usesCustomReplacement = hasCustomTermReplacement(terms);
|
|
if (!usesDefaultMask && !usesCustomReplacement) {
|
|
// The stored tree can be incomplete: GitHub truncates tree listings of
|
|
// very large repositories, and folders recorded in `truncatedFolders`
|
|
// have entries that never made it into the database. Ask GitHub
|
|
// directly for the path before concluding the file does not exist
|
|
// (#738). Without an anonymization mask the anonymized path is the
|
|
// original path, so it can be looked up as-is.
|
|
const recovered = await this.recoverTruncatedFile(fileDir);
|
|
if (recovered) {
|
|
this._file = recovered;
|
|
return recovered;
|
|
}
|
|
throw new AnonymousError("file_not_found", {
|
|
object: this,
|
|
httpStatus: 404,
|
|
});
|
|
}
|
|
|
|
// Custom replacements do not carry a marker that can be reversed into a
|
|
// narrow Mongo query. Fetch the repository's paths and verify them by
|
|
// re-applying anonymization. Default XXXX-N masks retain the optimized,
|
|
// anchored query.
|
|
const candidates = usesCustomReplacement
|
|
? await FileModel.find({ repoId: this.repository.repoId }).exec()
|
|
: await FileModel.find({
|
|
repoId: this.repository.repoId,
|
|
path: defaultMaskCandidateRegex(fileDir),
|
|
name: defaultMaskCandidateRegex(filename),
|
|
}).exec();
|
|
|
|
for (const candidate of candidates) {
|
|
const candidatePath = join(candidate.path, candidate.name);
|
|
if (
|
|
anonymizePath(candidatePath, terms) == this.anonymizedPath
|
|
) {
|
|
this._file = candidate;
|
|
return candidate;
|
|
}
|
|
}
|
|
|
|
// If applying the configured terms does not alter the requested path, it
|
|
// may simply be absent from a truncated tree and can be recovered as-is.
|
|
if (anonymizePath(this.anonymizedPath, terms) === this.anonymizedPath) {
|
|
const recovered = await this.recoverTruncatedFile(fileDir);
|
|
if (recovered) {
|
|
this._file = recovered;
|
|
return recovered;
|
|
}
|
|
}
|
|
throw new AnonymousError("file_not_found", {
|
|
object: this,
|
|
httpStatus: 404,
|
|
});
|
|
}
|
|
|
|
// On-demand recovery for files missing from the database because the
|
|
// GitHub tree listing was truncated. Only paths under a recorded
|
|
// truncated folder qualify — everything else is a genuine miss.
|
|
private async recoverTruncatedFile(fileDir: string): Promise<IFile | null> {
|
|
const truncated = this.repository.model.truncatedFolders || [];
|
|
const isAffected = truncated.some(
|
|
(folder) =>
|
|
folder === "" || fileDir === folder || fileDir.startsWith(folder + "/")
|
|
);
|
|
if (!isAffected) return null;
|
|
const source = this.repository.source as {
|
|
fetchFileInfoFromPath?: (filePath: string) => Promise<IFile | null>;
|
|
};
|
|
if (typeof source.fetchFileInfoFromPath !== "function") return null;
|
|
const recovered = await source.fetchFileInfoFromPath(this.anonymizedPath);
|
|
if (!recovered) return null;
|
|
recovered.repoId = this.repository.repoId;
|
|
logger.info("recovered file from truncated tree", {
|
|
repoId: this.repository.repoId,
|
|
path: this.anonymizedPath,
|
|
});
|
|
try {
|
|
// Cache it so the next request is served from the database.
|
|
await FileModel.create(recovered);
|
|
} catch (error) {
|
|
logger.warn(
|
|
"failed to cache recovered file",
|
|
serializeError(error as Error)
|
|
);
|
|
}
|
|
return recovered;
|
|
}
|
|
|
|
/**
|
|
* De-anonymize the path
|
|
*
|
|
* @returns the origin relative path of the file
|
|
*/
|
|
async originalPath(): Promise<string> {
|
|
if (this.anonymizedPath == null) {
|
|
throw new AnonymousError("path_not_specified", {
|
|
object: this,
|
|
httpStatus: 400,
|
|
});
|
|
}
|
|
if (!this._file) {
|
|
this._file = await this.getFileInfo();
|
|
}
|
|
return join(this._file.path, this._file.name);
|
|
}
|
|
extension() {
|
|
const filename = basename(this._file?.name || this.anonymizedPath);
|
|
const extensions = filename.split(".").reverse();
|
|
return extensions[0].toLowerCase();
|
|
}
|
|
isImage() {
|
|
const extension = this.extension();
|
|
return [
|
|
"png",
|
|
"jpg",
|
|
"jpeg",
|
|
"gif",
|
|
"svg",
|
|
"ico",
|
|
"bmp",
|
|
"tiff",
|
|
"tif",
|
|
"webp",
|
|
"avif",
|
|
"heif",
|
|
"heic",
|
|
].includes(extension);
|
|
}
|
|
|
|
isFileSupported() {
|
|
const extension = this.extension();
|
|
if (!this.repository.options.pdf && extension == "pdf") {
|
|
return false;
|
|
}
|
|
if (!this.repository.options.image && this.isImage()) {
|
|
return false;
|
|
}
|
|
return true;
|
|
}
|
|
|
|
async content(): Promise<Readable> {
|
|
if (this.anonymizedPath.includes(config.ANONYMIZATION_MASK)) {
|
|
await this.originalPath();
|
|
}
|
|
if (this._file?.size && this._file?.size > config.MAX_FILE_SIZE) {
|
|
throw new AnonymousError("file_too_big", {
|
|
object: this,
|
|
httpStatus: 413,
|
|
});
|
|
}
|
|
const content = await this.repository.source?.getFileContent(this);
|
|
const cacheWasReset = this.repository.model.isReseted;
|
|
if (cacheWasReset) {
|
|
await this.repository.markCachePresent();
|
|
}
|
|
if (cacheWasReset || this.repository.status != RepositoryStatus.READY) {
|
|
await this.repository.updateStatus(RepositoryStatus.READY);
|
|
}
|
|
return content;
|
|
}
|
|
|
|
async anonymizedContent() {
|
|
const anonymizer = this.repository.generateAnonymizeTransformer(
|
|
await this.originalPath()
|
|
);
|
|
if (!config.STREAMER_ENTRYPOINT) {
|
|
// collect the content locally
|
|
const content = await this.content();
|
|
content.on("error", (err) => anonymizer.destroy(err));
|
|
return content.pipe(anonymizer);
|
|
}
|
|
|
|
// use the streamer service
|
|
return got.stream(join(config.STREAMER_ENTRYPOINT, "api"), {
|
|
method: "POST",
|
|
json: {
|
|
token: await githubTokenForStreamer(await this.repository.getToken(), this.repository.model.source.repositoryName),
|
|
repoFullName: this.repository.model.source.repositoryName,
|
|
commit: this.repository.model.source.commit,
|
|
branch: this.repository.model.source.branch,
|
|
repoId: this.repository.repoId,
|
|
filePath: this.filePath,
|
|
sha: await this.sha(),
|
|
size: await this.size(),
|
|
anonymizerOptions: anonymizer.opt,
|
|
},
|
|
});
|
|
}
|
|
|
|
get filePath() {
|
|
if (!this._file) {
|
|
if (this.anonymizedPath.includes(config.ANONYMIZATION_MASK)) {
|
|
throw new AnonymousError("path_not_defined", {
|
|
object: this,
|
|
httpStatus: 400,
|
|
});
|
|
}
|
|
return this.anonymizedPath;
|
|
}
|
|
|
|
return join(this._file.path, this._file.name);
|
|
}
|
|
|
|
async send(res: Response): Promise<void> {
|
|
const anonymizer = this.repository.generateAnonymizeTransformer(
|
|
await this.originalPath()
|
|
);
|
|
// eslint-disable-next-line no-async-promise-executor
|
|
return new Promise<void>(async (resolve, reject) => {
|
|
try {
|
|
if (config.STREAMER_ENTRYPOINT) {
|
|
// use the streamer service
|
|
const [sha, size, token] = await Promise.all([
|
|
this.sha(),
|
|
this.size(),
|
|
this.repository.getToken(),
|
|
]);
|
|
const resStream = got
|
|
.stream(join(config.STREAMER_ENTRYPOINT, "api"), {
|
|
method: "POST",
|
|
json: {
|
|
sha,
|
|
size,
|
|
token: await githubTokenForStreamer(token, this.repository.model.source.repositoryName),
|
|
repoFullName: this.repository.model.source.repositoryName,
|
|
commit: this.repository.model.source.commit,
|
|
branch: this.repository.model.source.branch,
|
|
repoId: this.repository.repoId,
|
|
filePath: this.filePath,
|
|
anonymizerOptions: anonymizer.opt,
|
|
},
|
|
})
|
|
.on("error", (err: Error) => {
|
|
const { error } = streamerErrorToAnonymous(
|
|
err as Error & {
|
|
response?: { statusCode?: number; body?: unknown };
|
|
},
|
|
{
|
|
repoId: this.repository.repoId,
|
|
filePath: this.anonymizedPath,
|
|
}
|
|
);
|
|
error.value = this;
|
|
handleError(error, res);
|
|
});
|
|
// Forward Content-Type from the streamer's upstream response.
|
|
// got.stream(...).pipe(res) forwards body bytes only — without
|
|
// this, the parent response has no Content-Type and the browser
|
|
// guesses (text renders as download, images as octet-stream).
|
|
resStream.on("response", (upstream: { headers: Record<string, string | string[] | undefined> }) => {
|
|
if (res.headersSent) return;
|
|
const ct = upstream.headers["content-type"];
|
|
if (typeof ct === "string") {
|
|
res.contentType(ct);
|
|
} else {
|
|
const fallback = lookup(this.anonymizedPath);
|
|
if (fallback) res.contentType(fallback);
|
|
else if (isTextFile(this.anonymizedPath)) res.contentType("text/plain");
|
|
}
|
|
});
|
|
resStream.pipe(res);
|
|
// Resolve as soon as the response is fully written rather than
|
|
// waiting for the socket to close — keep-alive sockets stay open
|
|
// long after the body is delivered, and we don't want to delay
|
|
// post-send work like countView() that long.
|
|
res.on("finish", () => {
|
|
resolve();
|
|
});
|
|
res.on("close", () => {
|
|
resolve();
|
|
});
|
|
res.on("error", (err) => {
|
|
reject(err);
|
|
});
|
|
return;
|
|
}
|
|
|
|
const mime = lookup(this.anonymizedPath);
|
|
if (mime && this.extension() != "ts") {
|
|
res.contentType(mime);
|
|
} else if (isTextFile(this.anonymizedPath)) {
|
|
res.contentType("text/plain");
|
|
}
|
|
// For text files we anonymize on the fly and the output length can
|
|
// differ from the upstream, so byte ranges aren't meaningful — keep
|
|
// Accept-Ranges: none. For binary files (images, video, archives)
|
|
// the transformer is a passthrough, so omitting the explicit "none"
|
|
// lets <video>/<audio> elements use the standard fallback to a full
|
|
// download instead of refusing to play (#538).
|
|
const isTextEntry = isTextFile(this.anonymizedPath) === true;
|
|
if (isTextEntry) {
|
|
res.header("Accept-Ranges", "none");
|
|
}
|
|
anonymizer.once("transform", (data) => {
|
|
if (!mime && data.isText) {
|
|
res.contentType("text/plain");
|
|
}
|
|
});
|
|
const content = await this.content();
|
|
function handleStreamError(error: Error) {
|
|
if (!content.closed && !content.destroyed) {
|
|
content.destroy();
|
|
}
|
|
reject(error);
|
|
}
|
|
content
|
|
.on("error", handleStreamError)
|
|
.pipe(anonymizer)
|
|
.on("error", handleStreamError)
|
|
.pipe(res)
|
|
.on("error", handleStreamError)
|
|
.on("finish", () => {
|
|
// resolve on body fully written rather than waiting for the
|
|
// socket to close — keep-alive can hold the socket open long
|
|
// after the response is delivered, delaying post-send work.
|
|
if (!content.closed && !content.destroyed) {
|
|
content.destroy();
|
|
}
|
|
resolve();
|
|
})
|
|
.on("close", () => {
|
|
if (!content.closed && !content.destroyed) {
|
|
content.destroy();
|
|
}
|
|
resolve();
|
|
});
|
|
} catch (error) {
|
|
reject(error);
|
|
}
|
|
});
|
|
}
|
|
}
|