-
Notifications
You must be signed in to change notification settings - Fork 16.9k
Expand file tree
/
Copy pathfetch-logpush-datasets.ts
More file actions
187 lines (163 loc) · 5.62 KB
/
Copy pathfetch-logpush-datasets.ts
File metadata and controls
187 lines (163 loc) · 5.62 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
#!/usr/bin/env tsx
import fs from "fs";
import { createHash } from "node:crypto";
import { join } from "path";
import YAML from "yaml";
import {
downloadToDotTempIfNotPresent,
extractTarGz,
getDotTmpPath,
} from "../src/util/custom-loaders";
const MIDDLECACHE_BASE_URL = `${(
process.env.MIDDLECACHE_BASE_URL ?? "https://middlecache.ced.cloudflare.com"
).replace(/\/+$/, "")}/`;
const ARCHIVE_MIDDLECACHE_PATH = "v1/logpush-datasets/datasets.tar.gz";
const ARCHIVE_DOT_TMP_PATH = `middlecache/${ARCHIVE_MIDDLECACHE_PATH}`;
const DATASETS_DIR = "./src/content/docs/logs/logpush/logpush-job/datasets";
const EXTRACTED_DIR = join(".tmp", "logpush-datasets-extracted");
// --soft: warn and continue on failure instead of exiting non-zero.
// Used by the predev hook so a network failure doesn't block local development.
// --force: re-fetch even if the generated dataset pages exist, including a
// fresh download of the archive from middlecache.
const soft = process.argv.includes("--soft");
const force = process.argv.includes("--force");
const fail = (message: string): never => {
if (soft) {
const hasPages = fs.existsSync(DATASETS_DIR)
? getManagedScopes().some((scope) => getScopePages(scope).length > 0)
: false;
console.warn(
hasPages
? `Warning: ${message} — continuing with existing Logpush dataset pages`
: `Warning: ${message} — Logpush dataset pages are missing, /logs/logpush/logpush-job/datasets/ will not work`,
);
process.exit(0);
}
console.error(`Error: ${message}`);
process.exit(1);
};
// The scope dirs are hand-maintained (they carry index.mdx + sidebar wiring);
// generated .md pages are only synced into scopes that already exist.
const getManagedScopes = () =>
fs
.readdirSync(DATASETS_DIR, { withFileTypes: true })
.filter((entry) => entry.isDirectory())
.map((entry) => entry.name)
.filter((scope) => fs.existsSync(join(DATASETS_DIR, scope, "index.mdx")));
const getScopePages = (scope: string) => {
const scopeDir = join(DATASETS_DIR, scope);
return fs.existsSync(scopeDir)
? fs
.readdirSync(scopeDir)
.filter((file) => file.endsWith(".md"))
.sort()
: [];
};
const managedScopes = getManagedScopes();
const hasGeneratedPages =
fs.existsSync(DATASETS_DIR) &&
managedScopes.length > 0 &&
managedScopes.every((scope) => getScopePages(scope).length > 0);
if (hasGeneratedPages && !force) {
console.log(
"Logpush dataset pages already present, skipping fetch. (run `pnpm tsx bin/fetch-logpush-datasets.ts --force` to re-fetch)",
);
process.exit(0);
}
// Resolve the cache path the same way downloadToDotTempIfNotPresent does
// (repo-root `.tmp`, not cwd-relative) so the --force eviction always targets
// the file the downloader will reuse.
const archivePath = join(getDotTmpPath(), ...ARCHIVE_DOT_TMP_PATH.split("/"));
if (force) {
// --force means re-fetch from middlecache: drop the cached archive so
// downloadToDotTempIfNotPresent actually downloads rather than reusing it.
fs.rmSync(archivePath, { force: true });
}
console.log("Fetching Logpush dataset pages from middlecache");
try {
await downloadToDotTempIfNotPresent(
`${MIDDLECACHE_BASE_URL}${ARCHIVE_MIDDLECACHE_PATH}`,
ARCHIVE_DOT_TMP_PATH,
);
} catch (err) {
fail(`fetch failed: ${err}`);
}
const archiveSha256 = createHash("sha256")
.update(fs.readFileSync(archivePath))
.digest("hex");
console.log(`Logpush dataset archive SHA-256: ${archiveSha256}`);
// Remove any stale extracted content so we never sync pages from an old run.
fs.rmSync(EXTRACTED_DIR, { recursive: true, force: true });
try {
await extractTarGz(archivePath, EXTRACTED_DIR);
} catch (err) {
fail(`tar extraction failed: ${(err as Error).message}`);
}
const archiveScopes = fs
.readdirSync(EXTRACTED_DIR, { withFileTypes: true })
.filter((entry) => entry.isDirectory())
.map((entry) => entry.name);
// Warn about archive scopes that have no hand-written nav wiring yet; they are
// not synced until an index.mdx + sidebar entry are added for them.
for (const scope of archiveScopes) {
if (!managedScopes.includes(scope)) {
console.warn(
`Warning: skipping Logpush dataset scope not seeded in the docs: ${scope}`,
);
}
}
const pagesToCopy = new Map<string, string[]>();
for (const scope of managedScopes) {
const sourceDir = join(EXTRACTED_DIR, scope);
if (!fs.existsSync(sourceDir)) {
fail(
`Logpush dataset archive is missing scope: ${scope}. If intentional, remove that scope's generated pages in the same change`,
);
}
pagesToCopy.set(
scope,
fs
.readdirSync(sourceDir)
.filter((file) => file.endsWith(".md"))
.sort(),
);
}
// Validate frontmatter before touching the checked-in tree.
for (const [scope, pages] of pagesToCopy) {
for (const page of pages) {
const content = fs.readFileSync(join(EXTRACTED_DIR, scope, page), "utf8");
const frontmatter = /^---\r?\n([\s\S]*?)\r?\n---(?:$|\r?\n)/.exec(
content,
)?.[1];
const metadata = frontmatter
? (YAML.parse(frontmatter) as unknown)
: undefined;
if (
!metadata ||
typeof metadata !== "object" ||
!("title" in metadata) ||
typeof metadata.title !== "string" ||
metadata.title.trim() === ""
) {
fail(`Logpush dataset page has invalid frontmatter: ${scope}/${page}`);
}
}
}
let written = 0;
let removed = 0;
for (const [scope, pages] of pagesToCopy) {
const scopeDir = join(DATASETS_DIR, scope);
for (const page of getScopePages(scope)) {
if (!pages.includes(page)) {
fs.rmSync(join(scopeDir, page));
removed++;
}
}
for (const page of pages) {
fs.copyFileSync(join(EXTRACTED_DIR, scope, page), join(scopeDir, page));
written++;
}
}
console.log(
`Logpush dataset pages ready (${written} written, ${removed} removed)`,
);

