// ==UserScript== // @name Teams Transcript Extractor // @name:en Teams Transcript Extractor // @name:ja Teams 文字起こし抽出ツール // @name:zh-CN Teams 会议转录提取器 // @name:zh-TW Teams 會議轉錄提取器 // @name:ko Teams 회의 대화록 추출기 // @namespace https://github.com/CheerChen // @version 1.1.2 // @description Extract the full meeting transcript from a Teams recording's SharePoint Stream page via the page's own OneDrive transcript API. Works even when the download button is blocked. // @description:en Extract the full meeting transcript from a Teams recording's SharePoint Stream page via the page's own OneDrive transcript API. Works even when the download button is blocked. // @description:ja Teams録画のSharePoint Streamページから、ページ自身のOneDrive文字起こしAPI経由で全文を抽出します。ダウンロードボタンがブロックされていても動作します。 // @description:zh-CN 通过页面自身的 OneDrive 转录 API,从 Teams 录像的 SharePoint Stream 页面提取完整会议转录。即使下载按钮被禁用也可用。 // @description:zh-TW 透過頁面自身的 OneDrive 轉錄 API,從 Teams 錄影的 SharePoint Stream 頁面提取完整會議轉錄。即使下載按鈕被停用也可用。 // @description:ko 페이지 자체의 OneDrive 대화록 API를 통해 Teams 녹화의 SharePoint Stream 페이지에서 전체 회의 대화록을 추출합니다. 다운로드 버튼이 차단되어 있어도 작동합니다. // @author cheerchen37 // @match *://*/_layouts/15/stream.aspx* // @match *://*/*/_layouts/15/stream.aspx* // @match *://*/_layouts/15/streamembed.aspx* // @match *://*/*/_layouts/15/streamembed.aspx* // @match *://*/_layouts/15/xplatplugins.aspx* // @match *://*/*/_layouts/15/xplatplugins.aspx* // @grant unsafeWindow // @run-at document-idle // @icon https://www.google.com/s2/favicons?domain=teams.microsoft.com // @license MIT // @homepage https://github.com/CheerChen/userscripts // @supportURL https://github.com/CheerChen/userscripts/issues // @updateURL https://raw.githubusercontent.com/CheerChen/userscripts/master/teams-transcript-extractor.user.js // ==/UserScript== // The transcript lives in SharePoint-hosted documents: stream.aspx (recording page), // streamembed.aspx + xplatplugins.aspx (the two OOPIFs inside a Teams recap). // Verified 2026-09: the recap iframes load from -my.sharepoint.com under a // /personal// prefix, and the Teams top page cannot read the iframe URL // (src attr unset, contentWindow cross-origin) — so matching the _layouts path on // any host / any prefix is required. The g_fileInfo guard no-ops elsewhere. (function () { 'use strict'; const PAGE = typeof unsafeWindow !== 'undefined' ? unsafeWindow : window; const TAG = '[Teams Transcript]'; const log = (...a) => console.log(TAG, ...a); const sleep = (ms) => new Promise(r => setTimeout(r, ms)); // ---------- extraction method 1: OneDrive transcript API ---------- function itemBase() { const fi = PAGE.g_fileInfo; const u = fi && fi['.spItemUrl']; return u ? u.replace('/_api/v2.0/', '/_api/v2.1/').split('?')[0] : null; } // "00:12:34.5678" -> seconds function offsetToSec(s) { const m = /^(\d+):(\d+):(\d+(?:\.\d+)?)$/.exec(s || ''); return m ? +m[1] * 3600 + +m[2] * 60 + +m[3] : null; } async function extractViaApi() { const base = itemBase(); if (!base) throw new Error('g_fileInfo.spItemUrl not found (not a Stream recording page)'); const r1 = await fetch(base + '?select=media/transcripts&$expand=media/transcripts'); if (!r1.ok) throw new Error('transcript list HTTP ' + r1.status); const list = ((await r1.json()).media || {}).transcripts || []; if (!list.length) throw new Error('recording has no transcripts'); const t = list.find(x => x.isDefault) || list[0]; const r2 = await fetch(base + '/media/transcripts/' + t.id + '/streamContent?format=json'); if (!r2.ok) throw new Error('streamContent HTTP ' + r2.status); const data = await r2.json(); const speech = (data.entries || []).map(e => ({ kind: 'speech', sec: offsetToSec(e.startOffset), speaker: e.speakerDisplayName || '', text: e.text || '', speakerId: e.speakerId || null, roomId: e.roomId || null, })); const events = (data.events || []).map(e => ({ kind: 'system', sec: offsetToSec(e.startOffset), speaker: '', text: `${e.eventType}${e.userDisplayName ? ' by ' + e.userDisplayName : ''}`, eventType: e.eventType, user: e.userDisplayName || null, })); // API entries are already in display order; events slot in by offset. const entries = [...speech, ...events] .map((e, i) => ({ ...e, i })) .sort((a, b) => (a.sec ?? 0) - (b.sec ?? 0) || a.i - b.i) .map(({ i, ...e }) => e); log(`api: ${speech.length} utterances, ${events.length} events (transcript ${t.id}, ${list.length} available)`); return { method: 'api', entries, complete: true, transcripts: list.map(x => ({ id: x.id, displayName: x.displayName, languageTag: x.languageTag, isDefault: x.isDefault, source: x.source })), used: t.id, }; } // ---------- extraction method 2: DOM scroll harvest (fallback) ---------- // aria-label = "