Merge pull request #11 from Monadical-SAS/local-only

Local-only script
2026-02-04 18:06:48 +00:00 · 2023-06-14 22:45:58 +05:30
parent c8d4c179da 47915820fe
commit 088c0a224b
22 changed files with 345 additions and 0 deletions
--- a/.DS_Store
+++ b/.DS_Store
--- a/.vscode/settings.json
+++ b/.vscode/settings.json
@@ -0,0 +1,6 @@
 {
    "workbench.colorCustomizations": {
        "minimap.background": "#00000000",
        "scrollbar.shadow": "#00000000"
    }
 }
--- a/reflector-local/.DS_Store
+++ b/reflector-local/.DS_Store
--- a/reflector-local/0-reflector-local.py
+++ b/reflector-local/0-reflector-local.py
@@ -0,0 +1,33 @@
 import os
 import subprocess
 import sys
 from loguru import logger
 # Get the input file name from the command line argument
 input_file = sys.argv[1]  
 # example use: python 0-reflector-local.py input.m4a agenda.txt
 # Get the agenda file name from the command line argument if provided
 if len(sys.argv) > 2:
    agenda_file = sys.argv[2]
 else:
    agenda_file = "agenda.txt"
 # example use: python 0-reflector-local.py input.m4a my_agenda.txt
 # Check if the agenda file exists
 if not os.path.exists(agenda_file):
    logger.error("agenda_file is missing")
 # Check if the input file is .m4a, if so convert to .mp4
 if input_file.endswith(".m4a"):
    subprocess.run(["ffmpeg", "-i", input_file, f"{input_file}.mp4"])
    input_file = f"{input_file}.mp4" 
 # Run the first script to generate the transcript
 subprocess.run(["python3", "1-transcript-generator.py", input_file, f"{input_file}_transcript.txt"])
 # Run the second script to compare the transcript to the agenda
 subprocess.run(["python3", "2-agenda-transcript-diff.py", agenda_file, f"{input_file}_transcript.txt"])
 # Run the third script to summarize the transcript
 subprocess.run(["python3", "3-transcript-summarizer.py", f"{input_file}_transcript.txt", f"{input_file}_summary.txt"])
--- a/reflector-local/1-transcript-generator.py
+++ b/reflector-local/1-transcript-generator.py
@@ -0,0 +1,57 @@
 import argparse
 import os
 import moviepy.editor
 from loguru import logger
 import whisper
 WHISPER_MODEL_SIZE = "base"
 def init_argparse() -> argparse.ArgumentParser:
    parser = argparse.ArgumentParser(
        usage="%(prog)s <LOCATION> <OUTPUT>",
        description="Creates a transcript of a video or audio file using the OpenAI Whisper model"
    )
    parser.add_argument("location", help="Location of the media file")
    parser.add_argument("output", help="Output file path")
    return parser
 def main():
    import sys
    sys.setrecursionlimit(10000)
    parser = init_argparse()
    args = parser.parse_args()
    media_file = args.location
    logger.info(f"Processing file: {media_file}")
    # Check if the media file is a valid audio or video file
    if os.path.isfile(media_file) and not media_file.endswith(('.mp3', '.wav', '.ogg', '.flac', '.mp4', '.avi', '.flv')):
        logger.error(f"Invalid file format: {media_file}")
        return
    # If the media file we just retrieved is an audio file then skip extraction step
    audio_filename = media_file
    logger.info(f"Found audio-only file, skipping audio extraction")
    audio = moviepy.editor.AudioFileClip(audio_filename)
    logger.info("Selected extracted audio")
    # Transcribe the audio file using the OpenAI Whisper model
    logger.info("Loading Whisper speech-to-text model")
    whisper_model = whisper.load_model(WHISPER_MODEL_SIZE)
    logger.info(f"Transcribing file: {media_file}")
    whisper_result = whisper_model.transcribe(media_file)
    logger.info("Finished transcribing file")
    # Save the transcript to the specified file.
    logger.info(f"Saving transcript to: {args.output}")
    transcript_file = open(args.output, "w")
    transcript_file.write(whisper_result["text"])
    transcript_file.close()
 if __name__ == "__main__":
    main()
--- a/reflector-local/2-agenda-transcript-diff.py
+++ b/reflector-local/2-agenda-transcript-diff.py
@@ -0,0 +1,64 @@
 import argparse
 import spacy
 from loguru import logger
 # Define the paths for agenda and transcription files
 def init_argparse() -> argparse.ArgumentParser:
    parser = argparse.ArgumentParser(
        usage="%(prog)s <AGENDA> <TRANSCRIPTION>",
        description="Compares the transcript of a video or audio file to an agenda using the SpaCy model"
    )
    parser.add_argument("agenda", help="Location of the agenda file")
    parser.add_argument("transcription", help="Location of the transcription file")
    return parser
 args = init_argparse().parse_args()
 agenda_path = args.agenda
 transcription_path = args.transcription
 # Load the spaCy model and add the sentencizer
 spaCy_model = "en_core_web_md"
 nlp = spacy.load(spaCy_model)
 nlp.add_pipe('sentencizer')
 logger.info("Loaded spaCy model " + spaCy_model )
 # Load the agenda
 with open(agenda_path, "r") as f:
    agenda = [line.strip() for line in f.readlines() if line.strip()]
 logger.info("Loaded agenda items")
 # Load the transcription
 with open(transcription_path, "r") as f:
    transcription = f.read()
 logger.info("Loaded transcription")
 # Tokenize the transcription using spaCy
 doc_transcription = nlp(transcription)
 logger.info("Tokenized transcription")
 # Find the items covered in the transcription
 covered_items = {}
 for item in agenda:
    item_doc = nlp(item)
    for sent in doc_transcription.sents:
        if not sent or not all(token.has_vector for token in sent):
            # Skip an empty span or one without any word vectors
            continue
        similarity = sent.similarity(item_doc)
        similarity_threshold = 0.7
        if similarity > similarity_threshold:  # Set the threshold to determine what is considered a match
            covered_items[item] = True
            break
 # Count the number of items covered and calculatre the percentage
 num_covered_items = sum(covered_items.values())
 percentage_covered = num_covered_items / len(agenda) * 100
 # Print the results
 print("💬 Agenda items covered in the transcription:")
 for item in agenda:
    if item in covered_items and covered_items[item]:
        print("✅ ", item)
    else:
        print("❌ ", item)
 print("📊 Coverage: {:.2f}%".format(percentage_covered))
 logger.info("Finished comparing agenda to transcription with similarity threshold of " + str(similarity_threshold))
--- a/reflector-local/3-transcript-summarizer.py
+++ b/reflector-local/3-transcript-summarizer.py
@@ -0,0 +1,86 @@
 import argparse
 import nltk
 nltk.download('stopwords')
 from nltk.corpus import stopwords
 from nltk.tokenize import word_tokenize, sent_tokenize
 from heapq import nlargest
 from loguru import logger
 # Function to initialize the argument parser
 def init_argparse():
    parser = argparse.ArgumentParser(
        usage="%(prog)s <TRANSCRIPT> <SUMMARY>",
        description="Summarization"
    )
    parser.add_argument("transcript", type=str, default="transcript.txt", help="Path to the input transcript file")
    parser.add_argument("summary", type=str, default="summary.txt", help="Path to the output summary file")
    parser.add_argument("--num_sentences", type=int, default=5, help="Number of sentences to include in the summary")
    return parser
 # Function to read the input transcript file
 def read_transcript(file_path):
    with open(file_path, "r") as file:
        transcript = file.read()
    return transcript
 # Function to preprocess the text by removing stop words and special characters
 def preprocess_text(text):
    stop_words = set(stopwords.words('english'))
    words = word_tokenize(text)
    words = [w.lower() for w in words if w.isalpha() and w.lower() not in stop_words]
    return words
 # Function to score each sentence based on the frequency of its words and return the top sentences
 def summarize_text(text, num_sentences):
    # Tokenize the text into sentences
    sentences = sent_tokenize(text)
    # Preprocess the text by removing stop words and special characters
    words = preprocess_text(text)
    # Calculate the frequency of each word in the text
    word_freq = nltk.FreqDist(words)
    # Calculate the score for each sentence based on the frequency of its words
    sentence_scores = {}
    for i, sentence in enumerate(sentences):
        sentence_words = preprocess_text(sentence)
        for word in sentence_words:
            if word in word_freq:
                if i not in sentence_scores:
                    sentence_scores[i] = word_freq[word]
                else:
                    sentence_scores[i] += word_freq[word]
    # Select the top sentences based on their scores
    top_sentences = nlargest(num_sentences, sentence_scores, key=sentence_scores.get)
    # Sort the top sentences in the order they appeared in the original text
    summary_sent = sorted(top_sentences)
    summary = [sentences[i] for i in summary_sent]
    return " ".join(summary)
 def main():
    # Initialize the argument parser and parse the arguments
    parser = init_argparse()
    args = parser.parse_args()
    # Read the input transcript file
    logger.info(f"Reading transcript from: {args.transcript}")
    transcript = read_transcript(args.transcript)
    # Summarize the transcript using the nltk library
    logger.info("Summarizing transcript")
    summary = summarize_text(transcript, args.num_sentences)
    # Write the summary to the output file
    logger.info(f"Writing summary to: {args.summary}")
    with open(args.summary, "w") as f:
        f.write("Summary of: " + args.transcript + "\n\n")
        f.write(summary)
    logger.info("Summarization completed")
 if __name__ == "__main__":
    main()
--- a/reflector-local/30min-CyberHR/30min-CyberHR-agenda.txt
+++ b/reflector-local/30min-CyberHR/30min-CyberHR-agenda.txt
@@ -0,0 +1,4 @@
 # Deloitte HR @ NYS Cybersecurity Conference
 - ways to retain and grow your workforce
 - how to enable cybersecurity professionals to do their best work
 - low-budget activities that can be implemented starting tomorrow
--- a/reflector-local/30min-CyberHR/30min-CyberHR-transcript.txt
+++ b/reflector-local/30min-CyberHR/30min-CyberHR-transcript.txt
--- a/reflector-local/30min-CyberHR/30min-CyberHR.m4a
+++ b/reflector-local/30min-CyberHR/30min-CyberHR.m4a
--- a/reflector-local/30min-CyberHR/30min-CyberHR.m4a.mp4_summary.txt
+++ b/reflector-local/30min-CyberHR/30min-CyberHR.m4a.mp4_summary.txt
@@ -0,0 +1,3 @@
 Summary of: 30min-CyberHR/30min-CyberHR.m4a.mp4_transcript.txt
 Since the workforce is an organization's most valuable asset, investing in workforce experience activities, we've found has lead to more productive work, more efficient work, more innovative approaches to the work, and more engaged teams which ultimately results in better mission outcomes for your organization. And this one really focuses on not just pulsing a workforce once a year through an annual HR survey of, how do you really feel like, you know, what leadership considerations should we implement or, you know, how can we enhance the performance management process. We've just found that, you know, by investing in this and putting the workforce as, you know, the center part of what you invest in as an organization and leaders, it's not only about retention, talent, you know, the cyber workforce crisis, but people want to do work well and they're able to get more done and achieve more without you, you know, directly supervising and micromanaging or looking at everything because, you know, you know, you know, you're not going to be able to do anything. I hope there was a little bit of, you know, the landscape of the cyber workforce with some practical tips that you can take away for how to just think about, you know, improving the overall workforce experience and investing in your employees. So with this, you know, we know that all of you are in the trenches every day, you're facing this, you're living this, and we are just interested to hear from all of you, you know, just to start, like, what's one thing that has worked well in your organization in terms of enhancing or investing in the workforce experience?
--- a/reflector-local/30min-CyberHR/30min-CyberHR.m4a.mp4_transcript.txt
+++ b/reflector-local/30min-CyberHR/30min-CyberHR.m4a.mp4_transcript.txt
--- a/reflector-local/42min-StartupsTechTalk/42min-StartupsTechTalk-AGENDA-FULL.txt
+++ b/reflector-local/42min-StartupsTechTalk/42min-StartupsTechTalk-AGENDA-FULL.txt
@@ -0,0 +1,47 @@
 AGENDA:  Most important things to look for in a start up
 TAM: Make sure the market is sufficiently large than once they win they can get rewarded
 - Medium sized markets that should be winner take all can work
 - TAM needs to be realistic of direct market size
 Product market fit: Being in a good market with a product than can satisfy that market
 - Solves a problem
 - Builds a solution a customer wants to buy
 - Either saves the customer something (time/money/pain) or gives them something (revenue/enjoyment)
 Unit economics: Profit for delivering all-in cost must be attractive (% or $ amount)
 - Revenue minus direct costs
 - Raw input costs (materials, variable labour), direct cost of delivering and servicing the sale
 - Attractive as a % of sales so it can contribute to fixed overhead
 - Look for high incremental contribution margin
 LTV CAC: Life-time value (revenue contribution) vs cost to acquire customer must be healthy
 - LTV = Purchase value x number of purchases x customer lifespan
 - CAC = All-in costs of sales + marketing over number of new customer additions
 - Strong reputation leads to referrals leads to lower CAC. Want customers evangelizing product/service
 - Rule of thumb higher than 3
 Churn: Fits into LTV, low churn leads to higher LTV and helps keep future CAC down
 - Selling to replenish revenue every year is hard
 - Can run through entire customer base over time
 - Low churn builds strong net dollar retention
 Business: Must have sufficient barriers to entry to ward off copy-cats once established
 - High switching costs (lock-in)
 - Addictive
 - Steep learning curve once adopted (form of switching cost)
 - Two sided liquidity
 - Patents, IP, Branding
 - No hyper-scaler who can roll over you quickly
 - Scale could be a barrier to entry but works against most start-ups, not for them
 - Once developed, answer question: Could a well funded competitor starting up today easily duplicate this business or is it cheaper to buy the start up?
 Founders: Must be religious about their product. Believe they will change the world against all odds.
 - Just money in the bank is not enough to build a successful company. Just good tech not enough
 to build a successful company
 - Founders must be motivated to build something, not (all) about money. They would be doing
 this for free because they believe in it. Not looking for quick score
 - Founders must be persuasive. They will be asking others to sacrifice to make their dream come
 to life. They will need to convince investors this company can work and deserves funding.
 - Must understand who the customer is and what problem they are helping to solve.
 - Founders aren’t expected to know all the preceding points in this document but have an understanding of most of this, and be able to offer a vision.
--- a/reflector-local/42min-StartupsTechTalk/42min-StartupsTechTalk-AGENDA-HEADERS.txt
+++ b/reflector-local/42min-StartupsTechTalk/42min-StartupsTechTalk-AGENDA-HEADERS.txt
@@ -0,0 +1,8 @@
 AGENDA:  Most important things to look for in a start up
 TAM: Make sure the market is sufficiently large than once they win they can get rewarded
 Product market fit: Being in a good market with a product than can satisfy that market
 Unit economics: Profit for delivering all-in cost must be attractive (% or $ amount)
 LTV CAC: Life-time value (revenue contribution) vs cost to acquire customer must be healthy
 Churn: Fits into LTV, low churn leads to higher LTV and helps keep future CAC down
 Business: Must have sufficient barriers to entry to ward off copy-cats once established
 Founders: Must be religious about their product. Believe they will change the world against all odds.
--- a/reflector-local/42min-StartupsTechTalk/42min-StartupsTechTalk-Summary.txt
+++ b/reflector-local/42min-StartupsTechTalk/42min-StartupsTechTalk-Summary.txt
@@ -0,0 +1,10 @@
 Summary of: recordings/42min-StartupsTechTalk.mp4
 The speaker discusses their plan to launch an investment company, which will sit on a pool of cash raised from various partners and investors. They will take equity stakes in startups that they believe have the potential to scale and become successful. The speaker emphasizes the importance of investing in companies that have a large total addressable market (TAM) and good product-market fit. They also discuss the concept of unit economics and how it is important to ensure that the profit from selling a product or service outweighs the cost of producing it. The speaker encourages their team to keep an eye out for interesting startups and to send them their way if they come across any.
 The conversation is about the importance of unit economics, incremental margin, lifetime value, customer acquisition costs, churn, and barriers to entry in evaluating businesses for investment. The speaker explains that companies with good unit economics and high incremental contribution margins are ideal for investment. Lifetime value measures how much a customer will spend on a business over their entire existence, while customer acquisition costs measure the cost of acquiring a new customer. Churn refers to the rate at which customers leave a business, and businesses with low churn tend to have high lifetime values. High barriers to entry, such as high switching costs, can make it difficult for competitors to enter the market and kill established businesses.
 The speaker discusses various factors that can contribute to a company's success and create a competitive advantage. These include making the product addictive, having steep learning curves, creating two-sided liquidity for marketplaces, having patents or intellectual property, strong branding, and scale as a barrier to entry. The speaker also emphasizes the importance of founders having a plan to differentiate themselves from competitors and avoid being rolled over by larger companies. Additionally, the speaker mentions MasterCard and Visa as examples of companies that invented their markets, while Apple was able to build a strong brand despite starting with no developers or users.
 The speaker discusses the importance of founders in building successful companies, emphasizing that they must be passionate and believe in their product. They should also be charismatic and able to persuade others to work towards their vision. The speaker cites examples of successful CEOs such as Zuckerberg, Steve Jobs, Elon Musk, Bill Gates, Jeff Bezos, Travis Kalanick, and emphasizes that luck is also a factor in success. The speaker encourages listeners to have a critical eye when evaluating startups and to look for those with a clear understanding of their customers and the problem they are solving.
--- a/reflector-local/42min-StartupsTechTalk/42min-StartupsTechTalk-Transcript.txt
+++ b/reflector-local/42min-StartupsTechTalk/42min-StartupsTechTalk-Transcript.txt
--- a/reflector-local/42min-StartupsTechTalk/42min-StartupsTechTalk.mp4_summary.txt
+++ b/reflector-local/42min-StartupsTechTalk/42min-StartupsTechTalk.mp4_summary.txt
@@ -0,0 +1,3 @@
 Summary of: 42min-StartupsTechTalk/42min-StartupsTechTalk.mp4_transcript.txt
 If you had perfect knowledge, and you need like one more piece of advertising, drove like 0.2 customers in each customer generates, like let's say you wanted to completely maximize, you'd make it say your contribution margin, on incremental sales, is just over what you're spending on ad revenue. Like if you're, I don't know, well, let's see, I got like you don't really want to advertise a ton in the huge and everywhere, and then getting to ubiquitous, because you grab it, damage your brands, but just like an economic textbook theory, and be like, it'd be that basic math. And the table's like exactly, we're going to be really cautious to like be able to move in a year if we need to, but Google's goal is going to be giving away foundational models, lock everyone in, make them use Google Cloud, make them use Google Tools, and it's going to be very hard to switch off. Like if you were starting to develop Figma, you might say, okay, well Adobe is just gonna eat my lunch, right, like right away. So when you see a startup or talk to a founder and he's saying these things in your head like, man, this isn't gonna work because of, you know, there's no tab or there's, you know, like Amazon's gonna roll these cuts over in like two days or whatever, you know, or the man, this is really interesting because not only they're not doing it and no one else is doing this, but like they're going after a big market.
--- a/reflector-local/42min-StartupsTechTalk/42min-StartupsTechTalk.mp4_transcript.txt
+++ b/reflector-local/42min-StartupsTechTalk/42min-StartupsTechTalk.mp4_transcript.txt
--- a/reflector-local/7min-SmolDeveloper/7min-SmolDeveloper-AGENDA.txt
+++ b/reflector-local/7min-SmolDeveloper/7min-SmolDeveloper-AGENDA.txt
@@ -0,0 +1,4 @@
 GitHub
 Requirements
 Junior Developers
 Riding Elephants
--- a/reflector-local/7min-SmolDeveloper/7min-SmolDeveloper-Summary.txt
+++ b/reflector-local/7min-SmolDeveloper/7min-SmolDeveloper-Summary.txt
@@ -0,0 +1,4 @@
 Summary of: https://www.youtube.com/watch?v=DzRoYc2UGKI
 Small Developer is a program that creates an entire project for you based on a prompt. It uses the JATGPT API to generate code and files, and it's easy to use. The program can be installed by cloning the GitHub repository and using modalcom. The program can create projects for various languages, including Python and Ruby. You can also create a prompt.md file to input your prompt instead of pasting it into the terminal. The program is useful for creating detailed specs that can be passed on to junior developers. Overall, Small Developer is a helpful tool for quickly generating code and projects.
--- a/reflector-local/7min-SmolDeveloper/7min-SmolDeveloper-Transcript.txt
+++ b/reflector-local/7min-SmolDeveloper/7min-SmolDeveloper-Transcript.txt
--- a/reflector-local/readme.md
+++ b/reflector-local/readme.md
@@ -0,0 +1,11 @@
 # Record on Voice Memos on iPhone
 # Airdrop to MacBook Air
 # Run Reflector on .m4a Recording and Agenda
 python 0-reflector-local.py voicememo.m4a agenda.txt
 OR - using 30min-CyberHR example:
 python 0-reflector-local.py 30min-CyberHR/30min-CyberHR.m4a 30min-CyberHR/30min-CyberHR-agenda.txt
		`@@ -0,0 +1,3 @@`
							`Summary of: 30min-CyberHR/30min-CyberHR.m4a.mp4_transcript.txt`

							Since the workforce is an organization's most valuable asset, investing in workforce experience activities, we've found has lead to more productive work, more efficient work, more innovative approaches to the work, and more engaged teams which ultimately results in better mission outcomes for your organization. And this one really focuses on not just pulsing a workforce once a year through an annual HR survey of, how do you really feel like, you know, what leadership considerations should we implement or, you know, how can we enhance the performance management process. We've just found that, you know, by investing in this and putting the workforce as, you know, the center part of what you invest in as an organization and leaders, it's not only about retention, talent, you know, the cyber workforce crisis, but people want to do work well and they're able to get more done and achieve more without you, you know, directly supervising and micromanaging or looking at everything because, you know, you know, you know, you're not going to be able to do anything. I hope there was a little bit of, you know, the landscape of the cyber workforce with some practical tips that you can take away for how to just think about, you know, improving the overall workforce experience and investing in your employees. So with this, you know, we know that all of you are in the trenches every day, you're facing this, you're living this, and we are just interested to hear from all of you, you know, just to start, like, what's one thing that has worked well in your organization in terms of enhancing or investing in the workforce experience?
		`@@ -0,0 +1,3 @@`
							`Summary of: 42min-StartupsTechTalk/42min-StartupsTechTalk.mp4_transcript.txt`

							If you had perfect knowledge, and you need like one more piece of advertising, drove like 0.2 customers in each customer generates, like let's say you wanted to completely maximize, you'd make it say your contribution margin, on incremental sales, is just over what you're spending on ad revenue. Like if you're, I don't know, well, let's see, I got like you don't really want to advertise a ton in the huge and everywhere, and then getting to ubiquitous, because you grab it, damage your brands, but just like an economic textbook theory, and be like, it'd be that basic math. And the table's like exactly, we're going to be really cautious to like be able to move in a year if we need to, but Google's goal is going to be giving away foundational models, lock everyone in, make them use Google Cloud, make them use Google Tools, and it's going to be very hard to switch off. Like if you were starting to develop Figma, you might say, okay, well Adobe is just gonna eat my lunch, right, like right away. So when you see a startup or talk to a founder and he's saying these things in your head like, man, this isn't gonna work because of, you know, there's no tab or there's, you know, like Amazon's gonna roll these cuts over in like two days or whatever, you know, or the man, this is really interesting because not only they're not doing it and no one else is doing this, but like they're going after a big market.
		`@@ -0,0 +1,4 @@`
							`Summary of: https://www.youtube.com/watch?v=DzRoYc2UGKI`

							Small Developer is a program that creates an entire project for you based on a prompt. It uses the JATGPT API to generate code and files, and it's easy to use. The program can be installed by cloning the GitHub repository and using modalcom. The program can create projects for various languages, including Python and Ruby. You can also create a prompt.md file to input your prompt instead of pasting it into the terminal. The program is useful for creating detailed specs that can be passed on to junior developers. Overall, Small Developer is a helpful tool for quickly generating code and projects.