From 8cd36f576978c3e4b989e73f3b0341065bf44f1c Mon Sep 17 00:00:00 2001 From: Matt Williams Date: Sun, 30 Jun 2024 16:02:45 -0700 Subject: [PATCH] Add code for 2024-06-29 video Signed-off-by: Matt Williams --- 2024-06-30-usingfabric/.gitignore | 175 +++++++++++++++++++++++ 2024-06-30-usingfabric/README.md | 15 ++ 2024-06-30-usingfabric/bun.lockb | Bin 0 -> 6590 bytes 2024-06-30-usingfabric/index.ts | 205 +++++++++++++++++++++++++++ 2024-06-30-usingfabric/package.json | 17 +++ 2024-06-30-usingfabric/tsconfig.json | 27 ++++ 6 files changed, 439 insertions(+) create mode 100644 2024-06-30-usingfabric/.gitignore create mode 100644 2024-06-30-usingfabric/README.md create mode 100755 2024-06-30-usingfabric/bun.lockb create mode 100644 2024-06-30-usingfabric/index.ts create mode 100644 2024-06-30-usingfabric/package.json create mode 100644 2024-06-30-usingfabric/tsconfig.json diff --git a/2024-06-30-usingfabric/.gitignore b/2024-06-30-usingfabric/.gitignore new file mode 100644 index 0000000..9b1ee42 --- /dev/null +++ b/2024-06-30-usingfabric/.gitignore @@ -0,0 +1,175 @@ +# Based on https://raw.githubusercontent.com/github/gitignore/main/Node.gitignore + +# Logs + +logs +_.log +npm-debug.log_ +yarn-debug.log* +yarn-error.log* +lerna-debug.log* +.pnpm-debug.log* + +# Caches + +.cache + +# Diagnostic reports (https://nodejs.org/api/report.html) + +report.[0-9]_.[0-9]_.[0-9]_.[0-9]_.json + +# Runtime data + +pids +_.pid +_.seed +*.pid.lock + +# Directory for instrumented libs generated by jscoverage/JSCover + +lib-cov + +# Coverage directory used by tools like istanbul + +coverage +*.lcov + +# nyc test coverage + +.nyc_output + +# Grunt intermediate storage (https://gruntjs.com/creating-plugins#storing-task-files) + +.grunt + +# Bower dependency directory (https://bower.io/) + +bower_components + +# node-waf configuration + +.lock-wscript + +# Compiled binary addons (https://nodejs.org/api/addons.html) + +build/Release + +# Dependency directories + +node_modules/ +jspm_packages/ + +# Snowpack dependency directory (https://snowpack.dev/) + +web_modules/ + +# TypeScript cache + +*.tsbuildinfo + +# Optional npm cache directory + +.npm + +# Optional eslint cache + +.eslintcache + +# Optional stylelint cache + +.stylelintcache + +# Microbundle cache + +.rpt2_cache/ +.rts2_cache_cjs/ +.rts2_cache_es/ +.rts2_cache_umd/ + +# Optional REPL history + +.node_repl_history + +# Output of 'npm pack' + +*.tgz + +# Yarn Integrity file + +.yarn-integrity + +# dotenv environment variable files + +.env +.env.development.local +.env.test.local +.env.production.local +.env.local + +# parcel-bundler cache (https://parceljs.org/) + +.parcel-cache + +# Next.js build output + +.next +out + +# Nuxt.js build / generate output + +.nuxt +dist + +# Gatsby files + +# Comment in the public line in if your project uses Gatsby and not Next.js + +# https://nextjs.org/blog/next-9-1#public-directory-support + +# public + +# vuepress build output + +.vuepress/dist + +# vuepress v2.x temp and cache directory + +.temp + +# Docusaurus cache and generated files + +.docusaurus + +# Serverless directories + +.serverless/ + +# FuseBox cache + +.fusebox/ + +# DynamoDB Local files + +.dynamodb/ + +# TernJS port file + +.tern-port + +# Stores VSCode versions used for testing VSCode extensions + +.vscode-test + +# yarn v2 + +.yarn/cache +.yarn/unplugged +.yarn/build-state.yml +.yarn/install-state.gz +.pnp.* + +# IntelliJ based IDEs +.idea + +# Finder (MacOS) folder config +.DS_Store diff --git a/2024-06-30-usingfabric/README.md b/2024-06-30-usingfabric/README.md new file mode 100644 index 0000000..a1b306c --- /dev/null +++ b/2024-06-30-usingfabric/README.md @@ -0,0 +1,15 @@ +# 2024-06-30-usingfabric + +To install dependencies: + +```bash +bun install +``` + +To run: + +```bash +bun run index.ts +``` + +This project was created using `bun init` in bun v1.1.7. [Bun](https://bun.sh) is a fast all-in-one JavaScript runtime. diff --git a/2024-06-30-usingfabric/bun.lockb b/2024-06-30-usingfabric/bun.lockb new file mode 100755 index 0000000000000000000000000000000000000000..ae2c874dc734dbf98565b22a7420cdd3e8666e93 GIT binary patch literal 6590 zcmeHL2~<;O7JfiL4K5f##ZdtfR3tA8Yas;zcPxSxs0hUXFAxn0CNE*JgGkjuL_}~& zQAb6~I0dz+Bd7=W9dYc4y91&gP>WJ4ppG**_fPW6tD~46d(O<6<2~o)zJK|?d;jge zKiAQLr&cL>aZ)*tl56ba)N(dFT$v(%o^e6}e;5YD!}UsSs|Svrnh zlc?PvTE6m7`lv;|)s6!u2Amt?9;G5}p%aLeUJ*Ut&{>9ZDDI#u4~L?5JYdNcY$^w&#H&AHRxK<{XriF-52y;(3YU>p&uJ`kQqVr0$l(a z)n5jUc>Z9h2b;Qn2Lcm0O_yhe#Kp8`o7}G{oa(UHqVemZC9UI1t(Lx+6!7>)e$)4r zomG1lw_Pc7d0}UMqA8`4HM+U=_!zsxdk$9)^Uao9zsgnn2lZ%(-di9zXECNan7^oP zK}%w74(G=at9Vf>od5FaC%qJ+QxC(64{oi|rf-`2H|Zv)EZ4D1s5o%@g9NP#l5lvY z!T1m#sxy`sF~lgr z_&C6e4EBG`{zpd8T;us>YP64gpbj8Ng9#MNbi&lghd%Hi-47mwLje|*A*Sle13)7| z8nrXe*MHZ3XTMJ5&~GWgXF}+CnbfS1Cl&<*i}T9;dX!gR^cf$T2p*q;*lz zq!B~3CtSUb)Vf}>)qL~Ky~0(Kwry~Hz&U(nFgs+Z$mUV3ww=L?*B!adDoV^C?6>7y zal4tkWY6;zwU63%aMU?D8)om8$9BF_=C$Sy*Rlg<+1<&|ngm5ywyCY6%v_%5`D8y^ zt*OZ-`Z9Q78^0FThJ_LRr#t=5remV@>2}|3E7=zIFF;&Xs}W|3jaWX|}VBX?S;9fBP8D|$LD9Zqvr*W7rZxpn-^ z{$WWPw}UZTEp1|XLuF%HETVIwljk&-Odk1Scf-2LPgg8s@Z$5N)GyxGP(LnWgQ zE}vKSq>#ajZ#p{O`vHx0Q{taZT7C1xy5;vZqWzZwJ0yhxSI)X54f0>>6aE?R)8m0z zRSr%)ro<*Txd}uIrVifdF=+YDYcB_}>|CcZc;OE7T3EcI4H@-j^tr3@<*rwi``hm9 zyG7kyHoCa=aows54{zIMPKcTuy<=_LqVvgft{2)Z3k^PNW|ljlBq%3*NZsCMX9h1b zzOkyb>=j;-7o`KOG^!wh%z46}n;&~o@4FmcyV0P%|qhK{L)BHYKLvvKG}|++Aa#0rMZ$GD>LTJFdH>; zK|}KZ^DUEu_h#R(lE$}J5=RI5UelHpO=&)vb?ht4D}PkDFnG~70Sl|*?D6J0=^TIc zskE^Hrl$&~ADB^5=kJwEC5iK_i*M1}j(J{GE%Ym+lN`eMCrwrzSNW~fO5M%;)|#iz zj~ZLd;6-s03(Gk)&$oFC9mxJJY<^LRT|}03M3UR1lfLKu8{0K>nb-CTho;aqA+;lU z)r+;W(#@lJIbZEAba>WyIM_z=Fk>5om-+o-9ZKMYoxZWQ(B%iaxsUJNzY*T@pfRVk zA!VQP@4{EX5rvLD`*rT`2;2He_?F0T9das8&)(dV^Yzk&pB`Gk8hpE=jlqlJI~JD5 ziDXewxhO0mf9YAy)jLZAk0m%=ul$E=gV#SdisEPf{)cM!iu~`3wZ*ds^xrk7F>`O& zt@;9=rte~obUe+^5ibDVk7E1`;24Uv@yM>8AE&%)EKPui|#6j2i*bC_Zod$kvVkNK-7p9-IvE1_zR4%gZK~!;zo9m zJ!BWzL;T1NvWINEsnNF*eJ9Yj4Sk1@K4P$W#&BV0i)ITwW7OU-S)Q-QDd?IkoX5cV z1S^v-JxD&uCkT6B1JAN>Mxv*wOiWWMnjl=D1?5t3PQ%pXO93fB7@T#%*$-&plOCjy zAaGs==SNImUx0&v4P|a{21UmvBn4f$8=OOd7B|w96!Hn2t-)E7E)Q@%2j^9~T5!e( zXI^wI;2lBK!-T{CnJk=}F|>f!@4&eI))<^^!dV|v6EvX$FM`1NCY<*%HOcgb85Qse zoSDKIB4`ovNr8~`#TUcjD$QQbE`JP-iK3A;*=Op4d zr8J&SPoY%tN@)r`PWQ@#++@5WIYmuV+;|1_qqr(*vQ#EkO6l};g_>5!QCwOnmg_-D zG@YXI;qjDIqEtmI(@A+s@>~_EP$u&JvonwTmP)RGgsn`{i5cB+;j;k@_!V$&a16W0 z=}#M=^e);u+It*$=TrjQJKJDr5GxQf?vkLO z@vBTIhlH^PVtE1t8&ay^ij_)nI => { + let output = ""; + const rl = readline.createInterface({ + input: process.stdin, + output: process.stdout, + }); + + output = await rl.question("Enter the topic: "); + rl.close(); + return output; +}; + +const last6months = (): string => { + const now = new Date(); + const pastDate = new Date(now); + pastDate.setDate(now.getDate() - 180); + const formattedDate = pastDate.toISOString(); + return formattedDate; +}; + +// compare the video found to the query to ensure that its actually relevant +// I had gotten a bunch of irrelevant videos from the yt api +const confirmRelevance = async ( + query: string, + title: string, + description: string, +): Promise => { + const embeddedResults = await ollama.embeddings({ + model: "nomic-embed-text", + prompt: `Title: ${title},\n\n Description: ${description}`, + }); + const embeddedQuery = await ollama.embeddings({ + model: "nomic-embed-text", + prompt: query, + }); + + const relevance = + similarity(embeddedQuery.embedding, embeddedResults.embedding) || 0; + + return relevance; +}; + +// You can do this with the official api but it's harder. +const videoTranscript = async ( + videoid: string, + title: string, +): Promise => { + let transcript: TranscriptResponse[] = []; + try { + transcript = await YoutubeTranscript.fetchTranscript(videoid); + } catch (error) { + if (error === YoutubeTranscriptDisabledError) { + console.log(`Transcripts disabled on ${title}`); + } + } + return transcript; +}; + +const videoSearchTool = async ( + topic: string, + maxResults = 50, +): Promise => { + const results: VideoSearchResults[] = []; + const encodedTopic = encodeURIComponent(topic); + const apiKey = Bun.env.YouTubeAPIKey; + const url = `https://www.googleapis.com/youtube/v3/search?part=snippet&maxResults=${maxResults}&q=${encodedTopic}&type=video&key=${apiKey}&publishedAfter=${last6months()}`; + + const response = await fetch(url); + const data = await response.json(); + + if (data.items) { + for (const item of data.items) { + const id = item.id.videoId; + const title = item.snippet.title; + const channelId = item.snippet.channelId; + const channelTitle = item.snippet.channelTitle; + const description = item.snippet.description; + const daysSincePublished = Math.floor( + (new Date().getTime() - + new Date(item.snippet.publishedAt.slice(0, 10)).getTime()) / + (1000 * 60 * 60 * 24), + ); + const relevance = await confirmRelevance(topic, title, description); + if (relevance > 0.5) { + results.push({ + id, + title, + channelId, + channelTitle, + daysSincePublished, + description, + relevance, + }); + } + } + } + + return results; +}; + +const videoDetailsTool = async ( + video: VideoSearchResults, +): Promise => { + const apiKey = Bun.env.YouTubeAPIKey; + const videoUrl = `https://www.googleapis.com/youtube/v3/videos?part=snippet,statistics&id=${video.id}&key=${apiKey}`; + const channelUrl = `https://www.googleapis.com/youtube/v3/channels?part=statistics&id=${video.channelId}&key=${apiKey}`; + + const videoResponse = await fetch(videoUrl); + const channelResponse = await fetch(channelUrl); + + const videoData = await videoResponse.json(); + const channelData = await channelResponse.json(); + + const title = videoData.items[0].snippet.title; + const viewCount = videoData.items[0].statistics.viewCount as number; + const likeCount = videoData.items[0].statistics.likeCount; + const commentCount = videoData.items[0].statistics.commentCount; + const subscriberCount = channelData.items[0].statistics.subscriberCount; + const description = video.description; + const relevance = video.relevance; + const videoString = ` - Title: ${title}\n - Channel: ${video.channelTitle}\n - View Count: ${Number(viewCount).toLocaleString()}\n - Days Since Published: ${video.daysSincePublished}\n - Likes: ${Number(likeCount).toLocaleString()}\n - Subscriber Count: ${Number(subscriberCount).toLocaleString("en-US")}\n - Video URL: https://www.youtube.com/watch?v=${video.id}\n - Relevance: ${video.relevance}\n`; + + // const score = (viewCount / video.daysSincePublished / subscriberCount) + (likeCount * 10) + (commentCount * 100); + const score = + (viewCount / subscriberCount) * 0.4 + + (likeCount / viewCount) * 0.3 + + (commentCount / viewCount) * 0.2 * (1 / video.daysSincePublished); + + return { + videoString, + title, + viewCount, + likeCount, + commentCount, + subscriberCount, + score, + id: video.id, + description, + relevance, + }; +}; + + +await ollama.pull({ model: "m/llama3:8b-max" }); +await ollama.pull({ model: "nomic-embed-text" }); +const topic = await getTopic(); //just asks for the topic +const videos = await videoSearchTool(topic, 50); // get 50 results to play with +let scoredVideos: VideoDetails[] = []; +for (const video of videos) { + const videodetails = await videoDetailsTool(video); + scoredVideos.push(videodetails); +} + +scoredVideos = scoredVideos.sort((a, b) => b.score - a.score).slice(0, 5); +for (let index = 0; index < scoredVideos.length; index++) { + let report = "No Transcript Available"; + const video = scoredVideos[index]; + const transcript = await videoTranscript(video.id, video.title); + console.log(`Video ${index + 1}: ${video.score}`); + console.log(video.videoString); + if (transcript.length > 1) { + report = ( + await ollama.generate({ + model: "m/llama3:8b-max", + prompt: `${wisdomPrompt}\n\nTranscript: ${transcript.map((t) => t.text).join("")}`, + }) + ).response; + } + console.log(`${report} \n\n`); +} diff --git a/2024-06-30-usingfabric/package.json b/2024-06-30-usingfabric/package.json new file mode 100644 index 0000000..e06ccb7 --- /dev/null +++ b/2024-06-30-usingfabric/package.json @@ -0,0 +1,17 @@ +{ + "name": "2024-06-30-usingfabric", + "module": "index.ts", + "type": "module", + "devDependencies": { + "@types/bun": "latest" + }, + "peerDependencies": { + "typescript": "^5.0.0" + }, + "dependencies": { + "compute-cosine-similarity": "^1.1.0", + "ollama": "^0.5.2", + "readline": "^1.3.0", + "youtube-transcript": "^1.2.1" + } +} \ No newline at end of file diff --git a/2024-06-30-usingfabric/tsconfig.json b/2024-06-30-usingfabric/tsconfig.json new file mode 100644 index 0000000..238655f --- /dev/null +++ b/2024-06-30-usingfabric/tsconfig.json @@ -0,0 +1,27 @@ +{ + "compilerOptions": { + // Enable latest features + "lib": ["ESNext", "DOM"], + "target": "ESNext", + "module": "ESNext", + "moduleDetection": "force", + "jsx": "react-jsx", + "allowJs": true, + + // Bundler mode + "moduleResolution": "bundler", + "allowImportingTsExtensions": true, + "verbatimModuleSyntax": true, + "noEmit": true, + + // Best practices + "strict": true, + "skipLibCheck": true, + "noFallthroughCasesInSwitch": true, + + // Some stricter flags (disabled by default) + "noUnusedLocals": false, + "noUnusedParameters": false, + "noPropertyAccessFromIndexSignature": false + } +}