AI skills in use in my daily
You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
 
 

952 lines
40 KiB

#!/usr/bin/env node
// scrape-transcript.mjs — Capture a Teams/Stream meeting transcript from the UI.
//
// Strategy: drive a logged-in Edge (persistent profile) at the supplied URL,
// listen on every network response, and grab the first body that matches the
// WEBVTT signature (text/vtt content-type OR body starting with "WEBVTT").
// That's the same file the transcript pane renders from, so we don't need to
// know the exact backend endpoint — works for Stream-on-SharePoint, the older
// Microsoft Stream, the Teams web recap, and anything else that uses WebVTT.
//
// If no transcript pane opens automatically, in --headed mode the user can
// click it themselves and the network listener still catches the response.
//
// Live-join links (teams.microsoft.com/meet/..., /l/meetup-join/...,
// meet.microsoft.com/...) ALWAYS land on the pre-join/lobby screen, never the
// recap, even for meetings that already ended — there's no point waiting out
// the timeout on one. isLiveJoinUrl() detects these and auto-switches to the
// manual walk (navigate to teams.microsoft.com/v2/, wait for the user to open
// Recap → Transcript themselves) right away instead of after a wasted --wait.
//
// USAGE
// node scrape-transcript.mjs --url <url> [--out <dir>] [--subject <name>]
// [--headed] [--wait 45] [--tail-wait 45] [--debug]
// (--url pointing at a live-join link auto-upgrades to --manual behavior)
// --tail-wait: extra seconds spent on gentle recovery nudges (End+ArrowDown)
// if the DOM-scrape virtualized-list walk plateaus a few cues short of the
// transcript's reported total (aria-setsize). Default 45s; 0 disables.
//
// EXIT CODES
// 0 ok, 2 bad args, 3 no transcript captured before timeout, 4 page load failed.
import { chromium } from "playwright";
import { mkdirSync, writeFileSync, existsSync, rmSync, cpSync } from "node:fs";
import { dirname, join, resolve } from "node:path";
import { fileURLToPath } from "node:url";
import { tmpdir } from "node:os";
// Force synchronous (line-buffered) stderr/stdout on Windows pipes so progress
// messages appear in real time instead of being held in a buffer until exit.
if (process.stderr && process.stderr._handle && typeof process.stderr._handle.setBlocking === "function") {
try { process.stderr._handle.setBlocking(true); } catch {}
}
if (process.stdout && process.stdout._handle && typeof process.stdout._handle.setBlocking === "function") {
try { process.stdout._handle.setBlocking(true); } catch {}
}
const __dirname = dirname(fileURLToPath(import.meta.url));
// Canonical persistent profile (holds cookies, sign-in state). On Windows,
// ghost handles can leave this dir locked across runs even after the process
// that held it has exited. To make launches reliable we always clone this dir
// into a unique temp profile per run, launch from the clone, and copy cookies
// back at the end so the next run still benefits from cached auth.
const CANONICAL_PROFILE_DIR = join(__dirname, ".pw-profile");
// --- arg parsing -----------------------------------------------------------
function parseArgs(argv) {
const out = { headed: false, debug: false, waitSec: 45, tailWaitSec: 45 };
for (let i = 2; i < argv.length; i++) {
const a = argv[i];
const next = () => argv[++i];
switch (a) {
case "--url": out.url = next(); break;
case "--out": out.outDir = next(); break;
case "--subject": out.subject = next(); break;
case "--headed": out.headed = true; break;
case "--manual": out.manual = true; out.headed = true; break;
case "--wait": out.waitSec = Number.parseInt(next(), 10) || 45; break;
// Extra budget (default 45s) spent on gentle End/ArrowDown recovery
// nudges if the DOM-scrape virtualized-list walk plateaus a few cues
// short of the transcript's reported total. Set to 0 to disable.
case "--tail-wait": out.tailWaitSec = Number.parseInt(next(), 10); if (Number.isNaN(out.tailWaitSec)) out.tailWaitSec = 45; break;
// Testing/debug only: ignore any network VTT capture and force the DOM
// fallback path even when a network capture succeeded. Network capture
// availability is timing-dependent (Teams finishes processing the VTT
// within minutes of the meeting ending), which makes the DOM path hard
// to exercise on demand otherwise.
case "--force-dom": out.forceDom = true; break;
case "--debug": out.debug = true; break;
case "-h": case "--help": out.help = true; break;
default: throw new Error(`Unknown arg: ${a}`);
}
}
return out;
}
let args;
try {
args = parseArgs(process.argv);
} catch (e) {
process.stderr.write(`ERR: ${e.message}\n`);
process.exit(2);
}
if (args.help) {
process.stderr.write(`See header of ${import.meta.url}\n`);
process.exit(0);
}
if (!args.url && !args.manual) {
process.stderr.write("ERR: --url is required (or use --manual to start at teams.microsoft.com)\n");
process.exit(2);
}
if (args.manual && !args.url) {
args.url = "https://teams.microsoft.com/v2/";
}
// --- live-join link detection -----------------------------------------------
// Meeting-invite "Join" links (teams.microsoft.com/meet/..., /l/meetup-join/...,
// meet.microsoft.com/...) always land on the live pre-join/lobby screen, never
// on the recap — that's true for both live and already-ended meetings. There is
// no direct-URL path to a transcript from one of these, so waiting out the
// network/DOM-detection timeout on that URL is pure wasted time. Detect this
// case up front and immediately fall back to the manual walk instead of
// discovering it only after the full --wait timeout expires.
function isLiveJoinUrl(url) {
if (!url) return false;
return /teams\.microsoft\.com\/(l\/)?meetup-join\//i.test(url)
|| /teams\.microsoft\.com\/meet\//i.test(url)
|| /meet\.microsoft\.com\//i.test(url);
}
if (!args.manual && isLiveJoinUrl(args.url)) {
process.stderr.write(
`[info] "${args.url}" is a live-join link — these always open the pre-join\n` +
`screen, never the recap/transcript, even for meetings that already ended.\n` +
`Switching to manual mode automatically: opening Teams so you can navigate\n` +
`to Calendar → the meeting → Recap → Transcript yourself.\n`
);
args.manual = true;
args.autoSwitchedManual = true;
args.originalUrl = args.url;
args.url = "https://teams.microsoft.com/v2/";
}
// --- output paths ----------------------------------------------------------
function slugify(s) {
return (s || "meeting")
.toString()
.replace(/[\\/:*?"<>|]/g, "")
.replace(/\s+/g, " ")
.trim()
.slice(0, 80) || "meeting";
}
function todayISO() {
const d = new Date();
const pad = (n) => String(n).padStart(2, "0");
return `${d.getFullYear()}-${pad(d.getMonth() + 1)}-${pad(d.getDate())}`;
}
const subjectSlug = slugify(args.subject);
const outDir = resolve(
args.outDir || `D:\\Repos\\Obsidian\\02 - Meetings\\${todayISO()} - ${subjectSlug}`
);
mkdirSync(outDir, { recursive: true });
const vttPath = join(outDir, "transcript.vtt");
const mdPath = join(outDir, "transcript.md");
// --- WEBVTT helpers --------------------------------------------------------
function looksLikeVtt(text) {
if (!text) return false;
// Tolerate BOM + leading whitespace
const head = text.replace(/^\uFEFF/, "").trimStart();
return head.startsWith("WEBVTT");
}
function pickPreferredVtt(captures) {
// Prefer English; otherwise largest body (most cues).
const english = captures.filter((c) =>
/[-_./?&=](en|en[-_]us|en[-_]gb)\b/i.test(c.url) ||
/lang=en/i.test(c.url) ||
/\benglish\b/i.test(c.url)
);
const pool = english.length ? english : captures;
return pool.reduce((a, b) => (b.text.length > a.text.length ? b : a));
}
// Recursively walk a JSON value looking for an array of cue-shaped objects.
// Returns [{ start, end, speaker, text }] or null. Handles common shapes:
// Microsoft Stream: entries: [{ startOffset, endOffset, text, speakerDisplayName }]
// Teams/Stream new: items / cues / segments: [{ startTime, endTime, text, speakerId|speaker }]
// Bare arrays of such objects
function extractCuesFromJson(root) {
let best = null;
const visit = (node) => {
if (!node || typeof node !== "object") return;
if (Array.isArray(node)) {
if (node.length > 0 && typeof node[0] === "object" && node[0] !== null) {
const cues = node.map(cueFromObject).filter(Boolean);
if (cues.length >= 3 && (!best || cues.length > best.length)) best = cues;
}
for (const v of node) visit(v);
return;
}
for (const k of Object.keys(node)) visit(node[k]);
};
visit(root);
return best;
}
function cueFromObject(o) {
if (!o || typeof o !== "object") return null;
const text = o.text ?? o.transcript ?? o.caption ?? o.value ?? o.content;
if (typeof text !== "string" || !text.trim()) return null;
const start = parseTimeField(
o.startTime ?? o.startOffset ?? o.start ?? o.timeOffset ?? o.beginTime ?? o.startTimestamp
);
const end = parseTimeField(
o.endTime ?? o.endOffset ?? o.end ?? o.endTimestamp ?? (start != null ? start + 3 : null)
);
if (start == null) return null;
const speaker =
(o.speakerDisplayName || o.speakerName || o.speaker || o.author || o.userDisplayName || "")
.toString()
.trim();
return { start, end: end ?? start + 3, speaker, text: text.trim() };
}
function parseTimeField(v) {
if (v == null) return null;
if (typeof v === "number") {
// Heuristic: if >100000, assume ms; if also >100000000, assume ticks (100ns)
if (v > 100_000_000_000) return v / 10_000_000; // ticks
if (v > 100_000) return v / 1000; // ms
return v; // seconds
}
if (typeof v === "string") {
// ISO duration "PT1M23.45S"
const iso = v.match(/^PT(?:(\d+)H)?(?:(\d+)M)?(?:([\d.]+)S)?$/);
if (iso) return (parseInt(iso[1] || "0", 10) * 3600) + (parseInt(iso[2] || "0", 10) * 60) + parseFloat(iso[3] || "0");
// HH:MM:SS(.mmm) or MM:SS(.mmm)
const tm = v.match(/^(?:(\d+):)?(\d{1,2}):(\d{2}(?:\.\d+)?)$/);
if (tm) return (parseInt(tm[1] || "0", 10) * 3600) + (parseInt(tm[2], 10) * 60) + parseFloat(tm[3]);
const n = parseFloat(v);
if (!Number.isNaN(n)) return n;
}
return null;
}
function cuesToVtt(cues) {
const fmt = (s) => {
const ms = Math.floor((s % 1) * 1000);
const sec = Math.floor(s) % 60;
const min = Math.floor(s / 60) % 60;
const hr = Math.floor(s / 3600);
const pad = (n, w) => String(n).padStart(w, "0");
return `${pad(hr, 2)}:${pad(min, 2)}:${pad(sec, 2)}.${pad(ms, 3)}`;
};
let out = "WEBVTT\n\n";
for (const c of cues.sort((a, b) => a.start - b.start)) {
const tag = c.speaker ? `<v ${c.speaker}>` : "";
out += `${fmt(c.start)} --> ${fmt(Math.max(c.end, c.start + 0.1))}\n${tag}${c.text}\n\n`;
}
return out;
}
// WEBVTT cue parser: returns [{ start, end, speaker, text }]
function parseVtt(vtt) {
const lines = vtt.replace(/\r\n/g, "\n").split("\n");
const cues = [];
let i = 0;
// skip header
while (i < lines.length && !lines[i].includes("-->")) i++;
while (i < lines.length) {
// cue may have an id line above the timing line; back up if so
let timingLine = lines[i];
if (!timingLine || !timingLine.includes("-->")) { i++; continue; }
const m = timingLine.match(/(\d{1,2}:)?(\d{1,2}):(\d{2})(?:[.,](\d{1,3}))?\s*-->\s*(\d{1,2}:)?(\d{1,2}):(\d{2})(?:[.,](\d{1,3}))?/);
if (!m) { i++; continue; }
const toSec = (h, mm, ss, ms) =>
(parseInt(h || "0", 10) * 3600) + (parseInt(mm, 10) * 60) + parseInt(ss, 10) + (parseInt(ms || "0", 10) / 1000);
const start = toSec(m[1] && m[1].replace(":", ""), m[2], m[3], m[4]);
const end = toSec(m[5] && m[5].replace(":", ""), m[6], m[7], m[8]);
i++;
const textLines = [];
while (i < lines.length && lines[i].trim() !== "") {
textLines.push(lines[i]);
i++;
}
// skip blank
while (i < lines.length && lines[i].trim() === "") i++;
let raw = textLines.join(" ").trim();
if (!raw) continue;
// Speaker conventions: "<v Speaker Name>text</v>" OR "Speaker Name: text"
let speaker = "";
let text = raw;
const v = raw.match(/^<v\s+([^>]+)>(.*?)(?:<\/v>)?$/i);
if (v) {
speaker = v[1].trim();
text = v[2].trim();
} else {
const colon = raw.match(/^([A-Z][^:]{0,60}):\s+(.*)$/);
if (colon) { speaker = colon[1].trim(); text = colon[2].trim(); }
}
// strip remaining tags
text = text.replace(/<[^>]+>/g, "").trim();
cues.push({ start, end, speaker, text });
}
return cues;
}
function fmtTs(sec) {
const s = Math.max(0, Math.floor(sec));
const h = Math.floor(s / 3600);
const m = Math.floor((s % 3600) / 60);
const ss = s % 60;
const pad = (n) => String(n).padStart(2, "0");
return `${pad(h)}:${pad(m)}:${pad(ss)}`;
}
function cuesToMarkdown(cues, sourceUrl, outDirPath) {
// Extract date + title from the canonical "YYYY-MM-DD - <Title>" folder name.
// Falls back to today if the folder doesn't match the expected pattern.
const folderName = (outDirPath || "").split(/[\\\/]/).pop() || "";
const dateMatch = folderName.match(/^(\d{4}-\d{2}-\d{2})/);
const meetingDate = dateMatch ? dateMatch[1] : todayISO();
const titleMatch = folderName.match(/^\d{4}-\d{2}-\d{2}\s*-\s*(.+)$/);
const title = titleMatch ? titleMatch[1].trim() : (folderName || "Meeting");
const today = todayISO();
const out = [];
// Schema-conformant frontmatter per D:\Repos\Obsidian\AGENTS.md §3
out.push("---");
out.push(`type: meeting`);
out.push(`title: "${title.replace(/"/g, "'")}"`);
out.push(`date: ${meetingDate}`);
out.push(`created: ${today}`);
out.push(`updated: ${today}`);
out.push(`attendees: []`);
out.push(`tags: [meeting]`);
out.push(`source_url: "${sourceUrl}"`);
out.push(`captured_at: "${new Date().toISOString()}"`);
out.push(`cue_count: ${cues.length}`);
out.push("---");
out.push("");
out.push(`# ${title}`);
out.push("");
out.push("## Transcript");
out.push("");
// Group consecutive cues from same speaker
let i = 0;
while (i < cues.length) {
const speaker = cues[i].speaker || "Unknown";
const startTs = fmtTs(cues[i].start);
const parts = [cues[i].text];
let j = i + 1;
while (j < cues.length && (cues[j].speaker || "Unknown") === speaker) {
parts.push(cues[j].text);
j++;
}
out.push(`### ${startTs} — ${speaker}`);
out.push("");
out.push(parts.join(" ").replace(/\s+/g, " ").trim());
out.push("");
i = j;
}
return out.join("\n");
}
// --- DOM fallback: scrape Teams recap transcript pane (RecapxPlatIframe) ---
// The transcript lives in an iframe at *_layouts/15/xplatplugins.aspx*?hv=Recap
// inside a region with aria-label="Transcript". Each cue is a `.baseEntry-371`
// with aria-label="@N M minutes S seconds" and a `.entryText-372` child.
// Speaker headers (`.itemDisplayName-387`) appear only when speaker changes.
async function domScrapeFallback(page, debug, tailWaitMs) {
let recapFrame = null;
for (const f of page.frames()) {
const has = await f
.evaluate(() => !!document.querySelector('[aria-label="Transcript"][role="complementary"]'))
.catch(() => false);
if (has) { recapFrame = f; break; }
}
if (!recapFrame) {
process.stderr.write("[dom] no frame contains the transcript region\n");
return { vttText: null, cueCount: 0, targetSize: 0 };
}
process.stderr.write(`[dom] transcript frame found: ${recapFrame.url()}\n`);
// The recap transcript uses a virtual list. Generic scrollIntoView does not
// trigger more entries to render. The documented navigation is arrow keys —
// we focus the first cue and press ArrowDown repeatedly, harvesting between
// batches. Each ArrowDown moves selection forward, scrolling the virtual
// list naturally and rendering off-screen entries.
await recapFrame.evaluate(() => {
const region = document.querySelector('[aria-label="Transcript"][role="complementary"]');
if (!region) return;
const first = region.querySelector('[id^="entry-"] [data-is-focusable="true"]') ||
region.querySelector('[id^="entry-"]');
if (first && typeof first.focus === "function") first.focus();
});
const collected = new Map(); // entry-N -> cue
const seenSpeakers = []; // ordered list of speaker names by appearance
let lastSpeaker = "";
const harvestOnce = async () => {
const batch = await recapFrame.evaluate(() => {
const region = document.querySelector('[aria-label="Transcript"][role="complementary"]');
if (!region) return { entries: [], speakerHeaders: [] };
const entries = [];
for (const e of region.querySelectorAll('[id^="entry-"]')) {
const al = e.getAttribute("aria-label") || "";
const hM = al.match(/(\d+)\s*hours?/i);
const mM = al.match(/(\d+)\s*minutes?/i);
const sM = al.match(/(\d+)\s*seconds?/i);
let start = 0;
if (hM || mM || sM) {
start = (hM ? parseInt(hM[1], 10) * 3600 : 0) +
(mM ? parseInt(mM[1], 10) * 60 : 0) +
(sM ? parseInt(sM[1], 10) : 0);
}
const textEl = e.querySelector('[class*="entryText" i]') ||
e.querySelector('[id^="sub-entry-"]');
const text = (textEl ? textEl.innerText : e.innerText || "").trim();
if (!text) continue;
// Inline speaker from the row this entry belongs to
let inlineSpeaker = "";
const row = e.closest('[class*="listItemWithSpeaker" i]');
if (row) {
const dn = row.querySelector('[class*="itemDisplayName" i]');
if (dn) inlineSpeaker = (dn.innerText || "").trim();
}
entries.push({ id: e.id, idx: parseInt(e.id.slice("entry-".length), 10), start, text, inlineSpeaker });
}
// Also collect rendered speaker header positions (idx -> name) so we can
// forward-fill speakers for entries that don't have an inline header.
const speakerHeaders = [];
for (const row of region.querySelectorAll('[class*="listItemWithSpeaker" i]')) {
const dn = row.querySelector('[class*="itemDisplayName" i]');
const firstEntry = row.querySelector('[id^="entry-"]');
if (dn && firstEntry) {
speakerHeaders.push({
idx: parseInt(firstEntry.id.slice("entry-".length), 10),
name: (dn.innerText || "").trim(),
});
}
}
return { entries, speakerHeaders };
});
// Merge speaker headers
for (const sh of batch.speakerHeaders) {
seenSpeakers.push(sh);
}
for (const c of batch.entries) {
if (collected.has(c.id)) {
// Update speaker if previously empty
if (!collected.get(c.id).speaker && c.inlineSpeaker) {
collected.get(c.id).speaker = c.inlineSpeaker;
}
continue;
}
collected.set(c.id, {
idx: c.idx, start: c.start, speaker: c.inlineSpeaker || "", text: c.text,
});
}
};
// Read target
const targetSize = await recapFrame.evaluate(() => {
const probe = document.querySelector('[aria-label="Transcript"][role="complementary"] [aria-setsize]');
return probe ? parseInt(probe.getAttribute("aria-setsize") || "0", 10) : 0;
}).catch(() => 0);
await harvestOnce();
process.stderr.write(`[dom] initial harvest: ${collected.size}/${targetSize}\n`);
// Keyboard nav (End/ArrowDown) only moves list *focus*. Earlier revisions of
// this function tried dispatching a synthetic WheelEvent + manual scrollTop
// bump from inside the page — but testing showed the transcript region has
// no CSS overflow-scroll container at all (scrollHeight === clientHeight
// wherever we looked), so those synthetic events landed on nothing and
// never had a chance to help. Playwright's page.mouse.wheel() is different:
// it's a trusted, OS-level input event dispatched at real screen
// coordinates, which can reach handlers that reject synthetic/untrusted
// events (isTrusted checks) or rely on the browser's native wheel-to-scroll
// pipeline rather than a JS-visible overflow container. We position the
// mouse over the transcript pane first since wheel targets whatever is
// under the cursor.
const mouseWheelNudge = async (deltaY = 800) => {
try {
const handle = await recapFrame.$('[aria-label="Transcript"][role="complementary"]');
if (!handle) return { ok: false, reason: "no-region" };
const box = await handle.boundingBox();
if (!box) return { ok: false, reason: "no-bbox" };
const x = box.x + box.width / 2;
const y = box.y + box.height / 2;
await page.mouse.move(x, y);
await page.mouse.wheel(0, deltaY);
return { ok: true, x, y, deltaY };
} catch (e) {
return { ok: false, reason: e.message };
}
};
let stuck = 0;
let prevSize = collected.size;
const domStart = Date.now();
const domDeadlineMs = 600_000; // hard 10-min cap on DOM scrape
// Once we're this close to the target, the remaining rows are the tail of a
// virtualized list — big PageDown jumps are more likely to overshoot the
// render boundary there than in the middle of the transcript, which is how
// a run can plateau a few cues short of the end. Slow down and be patient
// once we cross this threshold.
const TAIL_FRACTION = 0.85;
const STUCK_LIMIT_NORMAL = 8;
const STUCK_LIMIT_TAIL = 24; // 3x more patience once we're in the tail zone
// Press ArrowDown in chunks; PageDown jumps faster when supported
for (let pass = 0; pass < 3000; pass++) {
const nearingEnd = targetSize > 0 && collected.size / targetSize >= TAIL_FRACTION;
const stuckLimit = nearingEnd ? STUCK_LIMIT_TAIL : STUCK_LIMIT_NORMAL;
if (stuck >= stuckLimit) break;
if (Date.now() - domStart > domDeadlineMs) {
process.stderr.write(`[dom] hard 10-min DOM scrape cap reached at pass=${pass} collected=${collected.size}/${targetSize}\n`);
break;
}
if (targetSize && collected.size >= targetSize) break;
if (nearingEnd) {
// Gentler single-step nudge with a longer settle wait so the last few
// virtualized rows have time to render before we check again. Every
// 4th stuck pass, throw in a real mouse-wheel nudge instead of a pure
// keyboard ArrowDown — some virtualized lists render more rows off a
// trusted scroll/wheel input rather than keyboard focus changes, so
// pure keyboard nav can plateau even when more content is available.
if (stuck > 0 && stuck % 4 === 0) {
const nudge = await mouseWheelNudge(800);
if (debug) process.stderr.write(`[dom] mouse wheel nudge (main loop): ${JSON.stringify(nudge)}\n`);
} else {
try { await page.keyboard.press("ArrowDown"); } catch {}
}
await page.waitForTimeout(275);
} else {
// Page down jumps faster; fall back to ArrowDown bursts.
try {
await page.keyboard.press("PageDown");
} catch {}
for (let k = 0; k < 5; k++) {
try { await page.keyboard.press("ArrowDown"); } catch {}
}
await page.waitForTimeout(150);
}
await harvestOnce();
if (collected.size === prevSize) stuck++;
else { stuck = 0; prevSize = collected.size; }
if (pass % 10 === 0) {
process.stderr.write(`[dom] pass=${pass} collected=${collected.size}/${targetSize} stuck=${stuck} tail=${nearingEnd} elapsed=${Math.round((Date.now() - domStart)/1000)}s\n`);
}
}
// One final End-key press in case there's still tail to load
try {
await page.keyboard.press("End");
await page.waitForTimeout(300);
await harvestOnce();
} catch {}
process.stderr.write(`[dom] main pass complete: ${collected.size}/${targetSize} cues in ${Math.round((Date.now() - domStart)/1000)}s\n`);
// Tail-recovery phase: if we're still short of the target after the main
// loop gave up, spend a bounded extra budget alternating techniques with
// generous settle waits: (1) keyboard End + ArrowDown nudges, and (2) a
// trusted mouse-wheel scroll positioned over the transcript pane. Some
// virtualized-list implementations render more rows off real, trusted
// scroll/wheel input rather than focus changes alone, so keyboard-only
// nudges can plateau even though more content is genuinely available. This
// is specifically for the case the main loop's stuck-counter trips right at
// the end of a virtualized list (observed: 307/329, 369/402, and 355/366
// captured, all plateauing short of the tail with keyboard nudges alone).
if (targetSize && collected.size < targetSize && tailWaitMs > 0) {
process.stderr.write(`[dom] tail recovery: ${collected.size}/${targetSize} — spending up to ${Math.round(tailWaitMs/1000)}s on recovery nudges\n`);
const recoveryDeadline = Date.now() + tailWaitMs;
let recoveryStuck = 0;
let recoveryPrevSize = collected.size;
let iteration = 0;
while (Date.now() < recoveryDeadline && collected.size < targetSize && recoveryStuck < 15) {
// Alternate: even iterations try keyboard, odd iterations try a real
// trusted mouse-wheel scroll. Doing both every iteration would conflate
// which technique is actually responsible for progress in the debug log.
if (iteration % 2 === 0) {
try { await page.keyboard.press("End"); } catch {}
await page.waitForTimeout(500);
try { await page.keyboard.press("ArrowDown"); } catch {}
await page.waitForTimeout(500);
} else {
const nudge = await mouseWheelNudge(800);
if (debug) process.stderr.write(`[dom] mouse wheel nudge (tail recovery): ${JSON.stringify(nudge)}\n`);
await page.waitForTimeout(500);
}
await harvestOnce();
if (collected.size === recoveryPrevSize) recoveryStuck++;
else { recoveryStuck = 0; recoveryPrevSize = collected.size; }
iteration++;
}
process.stderr.write(`[dom] tail recovery result: ${collected.size}/${targetSize}\n`);
}
process.stderr.write(`[dom] final harvest: ${collected.size}/${targetSize} cues in ${Math.round((Date.now() - domStart)/1000)}s\n`);
if (targetSize && collected.size < targetSize) {
process.stderr.write(
`[warn] transcript may be INCOMPLETE: captured ${collected.size} of ${targetSize} cues. ` +
`The missing cues are almost always a contiguous block (check the start/end of the transcript), not scattered.\n`
);
}
if (collected.size === 0) return { vttText: null, cueCount: 0, targetSize };
// Forward-fill speakers using seenSpeakers timeline
const cues = Array.from(collected.values()).sort((a, b) => a.idx - b.idx);
// Sort speaker headers by idx and forward-fill
const headers = seenSpeakers.slice().sort((a, b) => a.idx - b.idx);
let hi = 0;
let currentSpeaker = "";
for (const c of cues) {
while (hi < headers.length && headers[hi].idx <= c.idx) {
currentSpeaker = headers[hi].name;
hi++;
}
if (!c.speaker) c.speaker = currentSpeaker;
}
// Fill end times from next cue start; last cue gets +3s
for (let i = 0; i < cues.length; i++) {
cues[i].end = i + 1 < cues.length
? Math.max(cues[i + 1].start, cues[i].start + 0.5)
: cues[i].start + 3;
}
return { vttText: cuesToVtt(cues), cueCount: cues.length, targetSize };
}
// --- profile management ----------------------------------------------------
// Clone the canonical profile into a unique temp dir so we never collide with
// stale Windows file locks on the original. Cookies/auth state are copied via
// a best-effort sync back at the end.
function prepareProfileClone() {
const stamp = `${Date.now()}-${process.pid}`;
const cloneDir = join(tmpdir(), `pw-scrape-${stamp}`);
if (existsSync(CANONICAL_PROFILE_DIR)) {
process.stderr.write(`[profile] cloning ${CANONICAL_PROFILE_DIR} -> ${cloneDir}\n`);
try {
cpSync(CANONICAL_PROFILE_DIR, cloneDir, {
recursive: true,
force: true,
// Skip files that fail (e.g. locked lockfile) instead of aborting.
errorOnExist: false,
filter: (src) => !/[\\/](lockfile|SingletonLock|SingletonCookie|SingletonSocket)$/i.test(src),
});
} catch (e) {
process.stderr.write(`[profile] clone partial (${e.message}); continuing\n`);
}
} else {
process.stderr.write(`[profile] canonical profile missing; starting fresh (will need sign-in)\n`);
mkdirSync(cloneDir, { recursive: true });
}
return cloneDir;
}
function syncCookiesBack(cloneDir) {
// Best-effort: copy back the Default/Cookies* + Default/Network/Cookies* files
// so future runs benefit from refreshed auth. Skip silently on any failure.
try {
const dirsToSync = ["Default"];
for (const sub of dirsToSync) {
const srcSub = join(cloneDir, sub);
if (!existsSync(srcSub)) continue;
const dstSub = join(CANONICAL_PROFILE_DIR, sub);
try { mkdirSync(dstSub, { recursive: true }); } catch {}
const cookieNames = ["Cookies", "Cookies-journal", "Login Data", "Login Data-journal"];
for (const name of cookieNames) {
const s = join(srcSub, name);
if (!existsSync(s)) continue;
try { cpSync(s, join(dstSub, name), { force: true }); } catch {}
}
const networkDir = join(srcSub, "Network");
if (existsSync(networkDir)) {
const dstNetwork = join(dstSub, "Network");
try { mkdirSync(dstNetwork, { recursive: true }); } catch {}
for (const name of cookieNames) {
const s = join(networkDir, name);
if (!existsSync(s)) continue;
try { cpSync(s, join(dstNetwork, name), { force: true }); } catch {}
}
}
}
} catch (e) {
process.stderr.write(`[profile] cookie sync-back skipped: ${e.message}\n`);
}
}
// --- main ------------------------------------------------------------------
async function main() {
const profileDir = prepareProfileClone();
const ctx = await chromium.launchPersistentContext(profileDir, {
channel: "msedge",
headless: !args.headed,
viewport: { width: 1400, height: 900 },
args: ["--disable-blink-features=AutomationControlled"],
});
const page = ctx.pages()[0] || (await ctx.newPage());
const captures = []; // { url, text }
page.on("response", async (resp) => {
try {
const url = resp.url();
const ct = (resp.headers()["content-type"] || "").toLowerCase();
// Skip obvious non-text resources up front for speed
if (/\.(png|jpe?g|gif|webp|woff2?|ttf|css|mp4|mp3|m4s|ts|svg|ico)(\?|$)/i.test(url)) return;
if (ct.includes("image/") || ct.includes("font/") || ct.includes("video/") || ct.includes("audio/")) return;
if (ct.includes("javascript") || ct.includes("text/css") || ct.includes("text/html")) return;
const urlHints = /transcript|caption|texttrack|subtitle|\.vtt(\?|$)|stream\.aspx|mediastream|videostream|substrate.*media|recap/i.test(url);
const ctHints = ct.includes("text/vtt") || ct.includes("application/vtt") || ct.includes("application/json");
if (!urlHints && !ctHints) {
// Skip uninteresting responses without reading body (fast path)
return;
}
const cl = parseInt(resp.headers()["content-length"] || "0", 10);
if (cl > 0 && cl > 10_000_000) return;
let text = "";
try { text = await resp.text(); } catch { return; }
if (!text) return;
if (args.debug) {
const preview = text.slice(0, 120).replace(/\s+/g, " ");
process.stderr.write(`[debug] CANDIDATE ct=${ct} len=${text.length} url=${url}\n body[0:120]=${preview}\n`);
}
if (looksLikeVtt(text)) {
captures.push({ url, text, kind: "vtt" });
if (args.debug) process.stderr.write(`[debug] >>> CAPTURED native VTT (${text.length} bytes)\n`);
return;
}
// JSON transcript shapes: try to detect cue arrays
if (ct.includes("application/json") || /^[\s{[]/.test(text)) {
let json;
try { json = JSON.parse(text); } catch { return; }
const cues = extractCuesFromJson(json);
if (cues && cues.length > 0) {
captures.push({ url, text: cuesToVtt(cues), kind: "json", jsonCueCount: cues.length });
if (args.debug) process.stderr.write(`[debug] >>> CAPTURED JSON transcript (${cues.length} cues from ${url})\n`);
}
}
} catch (e) {
if (args.debug) process.stderr.write(`[debug] listener err: ${e.message}\n`);
}
});
try {
await page.goto(args.url, { waitUntil: "domcontentloaded", timeout: 60_000 });
} catch (e) {
process.stderr.write(`ERR: page load failed: ${e.message}\n`);
await ctx.close();
process.exit(4);
}
if (args.manual && args.waitSec < 300) {
args.waitSec = 300;
}
if (args.manual) {
const originHint = args.autoSwitchedManual
? ` (auto-switched from live-join link: ${args.originalUrl})\n`
: "";
process.stderr.write(
`\n[manual mode] Navigate to your meeting in the browser:\n` +
originHint +
` 1. Open the meeting from Calendar (or Chat)\n` +
` 2. Click 'Recap' tab → 'Transcript'\n` +
`Waiting up to ${args.waitSec}s for the transcript pane to load a .vtt file...\n\n`
);
}
// Manual mode no longer BLOCKS on stdin. The wait loop below auto-detects the
// transcript pane (the recap region + cue entries) as soon as you open it and
// proceeds on its own. Pressing Enter is an OPTIONAL manual override that
// forces extraction to start immediately, e.g. if auto-detect is slow.
let manualSignal = false;
if (args.manual) {
process.stderr.write(
`\n[manual mode] Navigate to the meeting Recap → Transcript pane, then click a\n` +
`transcript line so the list has focus. Extraction starts AUTOMATICALLY once\n` +
`the transcript pane is detected — you don't need to do anything else here.\n` +
`(Optional: press Enter in this terminal to force it to start immediately.)\n`
);
process.stdin.resume();
process.stdin.once("data", () => { manualSignal = true; });
}
// Try to auto-click a Transcript button across known UI variants and across
// all frames (Teams recap has the tab inside a SharePoint xplatplugins iframe).
const transcriptSelectors = [
'[data-tid="Transcript"]',
'button[aria-label*="Transcript" i]',
'button[title*="Transcript" i]',
'[role="tab"][aria-label*="Transcript" i]',
'[role="tab"]:has-text("Transcript")',
'button:has-text("Transcript")',
'div[role="button"]:has-text("Transcript")',
];
// Only auto-click a Transcript tab in direct-URL mode. In manual mode the user
// navigates to the recap themselves, so a one-shot click on whatever page is
// currently showing would be useless at best and misdirected at worst.
if (!args.manual) for (const frame of page.frames()) {
for (const sel of transcriptSelectors) {
try {
const loc = frame.locator(sel).first();
if (await loc.isVisible({ timeout: 500 }).catch(() => false)) {
await loc.click({ timeout: 2000 }).catch(() => {});
if (args.debug) process.stderr.write(`[debug] clicked transcript via ${sel} in ${frame.url()}\n`);
break;
}
} catch { /* try next */ }
}
}
// Wait for a network VTT capture OR the recap transcript pane to appear in any
// frame. Use the full wait window in all modes: in manual mode the loop needs
// time for you to navigate to the recap, but it breaks early the instant the
// transcript pane is detected (or you press Enter), so it's not a fixed delay.
const waitCap = args.waitSec;
const deadline = Date.now() + waitCap * 1000;
const startWait = Date.now();
let lastProgress = Date.now();
while (Date.now() < deadline) {
if (captures.length > 0 && !args.forceDom) {
process.stderr.write(`[info] network VTT captured after ${Math.round((Date.now() - startWait)/1000)}s\n`);
break;
}
if (args.manual && manualSignal) {
process.stderr.write(`[info] manual override received — starting extraction\n`);
await page.waitForTimeout(500);
break;
}
// Look for the transcript region in ANY frame (not just xplatplugins.aspx —
// Teams UI changes the iframe URL periodically).
let recapReady = false;
for (const f of page.frames()) {
try {
recapReady = await f
.evaluate(() => !!document.querySelector('[aria-label="Transcript"][role="complementary"] [id^="entry-"]'))
.catch(() => false);
if (recapReady) break;
} catch {}
}
if (recapReady) {
process.stderr.write(`[info] transcript pane DOM detected after ${Math.round((Date.now() - startWait)/1000)}s\n`);
await page.waitForTimeout(1500);
break;
}
if (Date.now() - lastProgress > 5000) {
process.stderr.write(`[info] still waiting for transcript pane (${Math.round((Date.now() - startWait)/1000)}s elapsed, cap=${waitCap}s, captures=${captures.length})\n`);
lastProgress = Date.now();
}
await page.waitForTimeout(500);
}
if (Date.now() >= deadline) {
process.stderr.write(`[info] wait window elapsed (${waitCap}s); proceeding with whatever's available (captures=${captures.length})\n`);
}
let vttText = null;
let source = null;
let domCueCount = null;
let domTargetSize = null;
if (captures.length > 0 && !args.forceDom) {
vttText = pickPreferredVtt(captures).text;
source = "network";
process.stderr.write(`[info] using network capture (${vttText.length} bytes)\n`);
} else {
if (args.forceDom && captures.length > 0) {
process.stderr.write(`[info] --force-dom set: ignoring ${captures.length} network capture(s), using DOM fallback\n`);
} else {
process.stderr.write("[info] no network VTT captured, trying DOM fallback\n");
}
const domResult = await domScrapeFallback(page, args.debug, args.tailWaitSec * 1000);
vttText = domResult.vttText;
domCueCount = domResult.cueCount;
domTargetSize = domResult.targetSize;
if (vttText) {
source = "dom";
process.stderr.write(`[info] DOM scrape produced ${vttText.length} bytes\n`);
}
}
// CRITICAL: write files BEFORE attempting browser cleanup. ctx.close() can
// hang on Windows when there are many in-flight listeners; we don't want to
// lose a captured transcript to a cleanup hang.
if (vttText) {
process.stderr.write(`[info] writing transcript files...\n`);
writeFileSync(vttPath, vttText, "utf8");
const cues = parseVtt(vttText);
const md = cuesToMarkdown(cues, args.url, outDir);
writeFileSync(mdPath, md, "utf8");
// Emit success JSON to stdout immediately so the caller knows we're done
// with the meaningful work, even if cleanup below takes a while.
const truncated = domTargetSize != null && domCueCount != null && domCueCount < domTargetSize;
process.stdout.write(JSON.stringify({
ok: true,
vtt: vttPath,
md: mdPath,
cueCount: cues.length,
targetCueCount: domTargetSize ?? undefined,
truncated,
bytes: Buffer.byteLength(vttText, "utf8"),
source,
outDir,
}) + "\n");
process.stderr.write(`[info] wrote ${cues.length} cues to ${vttPath}\n`);
if (truncated) {
process.stderr.write(
`[warn] transcript is likely INCOMPLETE (${domCueCount}/${domTargetSize} cues). ` +
`Re-run with a larger --tail-wait, or check the end of the transcript for the gap.\n`
);
}
}
// Best-effort browser cleanup with hard timeout so we never hang on close.
process.stderr.write(`[info] closing browser...\n`);
await Promise.race([
ctx.close().catch((e) => process.stderr.write(`[info] ctx.close warning: ${e.message}\n`)),
new Promise((resolve) => setTimeout(() => {
process.stderr.write(`[info] ctx.close timed out after 5s — moving on\n`);
resolve();
}, 5000)),
]);
// Sync cookies back so future runs benefit from refreshed auth.
process.stderr.write(`[info] syncing cookies back to canonical profile...\n`);
syncCookiesBack(profileDir);
// Best-effort cleanup of the temp profile clone.
try { rmSync(profileDir, { recursive: true, force: true }); } catch {}
if (!vttText) {
process.stderr.write(
`ERR: no transcript captured within ${args.waitSec}s. Re-run with --headed and open the transcript pane manually; or pass --debug to see network activity.\n`
);
process.exit(3);
}
// Force exit — sometimes lingering listeners or process refs keep node alive
// even after main() returns; we've already written the output and JSON.
process.stderr.write(`[info] done.\n`);
process.exit(0);
}
main().catch((e) => {
process.stderr.write(`ERR: ${e.stack || e.message}\n`);
process.exit(1);
});