Word count, sentences, top words and reading time for any paragraph.
/build/project-text-analyser with the starter and the rubric already in it.Analyse a paragraph: count words and sentences, find the longest word, estimate reading time, and list the five most used words once filler words like "the" and "is" are removed.
Every word counter, SEO tool and search engine starts with these same steps: clean the text, split it, count it. You will use regex, Counter and a two-part sort key, three things interviewers like to ask about.
You can clean text with regex, count with collections.Counter and sort with a tie-breaking key.
Each step is one function in text_stats.py. In the workspace, press Check next to a step: it runs your file and tells you exactly what is still wrong.
Build against this exact list — it is the spec your published project is judged on.
Copy this into a new file — or open it in a workspace as text_stats.py. The TODO parts are yours to fill.
"""
Paragraph text analyser.
Counts words and sentences, finds the most used words (skipping filler
words like "the"), the longest word and the reading time of a paragraph.
Fill in one function per step, press Run, then press "Check step".
"""
import math
import re
from collections import Counter
PARAGRAPH = """Python is a language that reads almost like English. Students in India
learn Python for placements, for data science and for automation. The best
way to learn Python is to write small programs every day! Write a program,
break the program, then fix the program. Is it slow at first? Yes, but the
habit of writing code daily beats reading about code."""
STOP_WORDS = {
"a", "an", "and", "are", "at", "for", "in", "is", "it", "like", "of",
"the", "then", "to", "but", "yes", "about", "every", "first",
}
def clean_words(text: str) -> list[str]:
# TODO: lowercase the text and pull out the words, dropping punctuation.
# Hint: re.findall(r"[a-z']+", text.lower())
raise NotImplementedError("step 1, clean_words(): text -> list of lowercase words")
def count_sentences(text: str) -> int:
# TODO: split on runs of . ! ? and count the pieces that are not blank.
# Hint: re.split(r"[.!?]+", text)
raise NotImplementedError("step 2, count_sentences(): count the sentences")
def longest_word(words: list[str]) -> str:
# TODO: the longest word; on a tie keep the first one. "" for no words.
raise NotImplementedError("step 3, longest_word(): the longest word")
def reading_time(words: list[str], wpm: int = 200) -> int:
# TODO: minutes to read at wpm words per minute, rounded UP, at least 1.
raise NotImplementedError("step 4, reading_time(): minutes, rounded up")
def word_frequencies(words: list[str], stop_words: set[str]) -> Counter:
# TODO: a Counter of the words that are NOT in stop_words.
raise NotImplementedError("step 5, word_frequencies(): count the useful words")
def top_words(freq: Counter, n: int) -> list[tuple[str, int]]:
# TODO: the n most frequent (word, count) pairs, highest count first,
# and alphabetical when two counts tie (so the output is stable).
raise NotImplementedError("step 6, top_words(): top n, ties alphabetical")
def main() -> None:
words = clean_words(PARAGRAPH)
print(f"Words: {len(words)}")
print(f"Sentences: {count_sentences(PARAGRAPH)}")
print(f"Unique words: {len(set(words))}")
print(f"Longest word: {longest_word(words)}")
print(f"Reading time: {reading_time(words)} min")
print("Top 5 words:")
for word, count in top_words(word_frequencies(words, STOP_WORDS), 5):
print(f" {word:<10}{count}")
if __name__ == "__main__":
try:
main()
except NotImplementedError as todo:
# A fresh starter is SUPPOSED to stop here. Say which step is next
# instead of printing a traceback that looks like a bug.
print(f"Not built yet: {todo}")
print("Write that function, then press Run again. Each step you finish moves this message forward.")
Try every step first. This version passes all 6 checks; yours can look different and still pass.
"""
Paragraph text analyser.
Counts words and sentences, finds the most used words (skipping filler
words like "the"), the longest word and the reading time of a paragraph.
"""
import math
import re
from collections import Counter
PARAGRAPH = """Python is a language that reads almost like English. Students in India
learn Python for placements, for data science and for automation. The best
way to learn Python is to write small programs every day! Write a program,
break the program, then fix the program. Is it slow at first? Yes, but the
habit of writing code daily beats reading about code."""
STOP_WORDS = {
"a", "an", "and", "are", "at", "for", "in", "is", "it", "like", "of",
"the", "then", "to", "but", "yes", "about", "every", "first",
}
def clean_words(text: str) -> list[str]:
return re.findall(r"[a-z']+", text.lower())
def count_sentences(text: str) -> int:
return len([s for s in re.split(r"[.!?]+", text) if s.strip()])
def word_frequencies(words: list[str], stop_words: set[str]) -> Counter:
return Counter(w for w in words if w not in stop_words)
def top_words(freq: Counter, n: int) -> list[tuple[str, int]]:
return sorted(freq.items(), key=lambda kv: (-kv[1], kv[0]))[:n]
def longest_word(words: list[str]) -> str:
return max(words, key=len) if words else ""
def reading_time(words: list[str], wpm: int = 200) -> int:
return max(1, math.ceil(len(words) / wpm))
def main() -> None:
words = clean_words(PARAGRAPH)
print(f"Words: {len(words)}")
print(f"Sentences: {count_sentences(PARAGRAPH)}")
print(f"Unique words: {len(set(words))}")
print(f"Longest word: {longest_word(words)}")
print(f"Reading time: {reading_time(words)} min")
print("Top 5 words:")
for word, count in top_words(word_frequencies(words, STOP_WORDS), 5):
print(f" {word:<10}{count}")
if __name__ == "__main__":
main()
text_stats.py and a README.md holding all 5 rubric items as a checklist.pyrun.in/u/<handle>/w/project-text-analyser.pyrun.in/u/<handle> portfolio page.Your workspace saves to this browser only. Sign in to keep it across devices and to publish it.
Flip the workspace to Public and paste its URL below, or push the file to a GitHub gist or repo. Paste the code inline too if you want the optional AI review to comment on specific lines.
A published PyRun workspace URL (pyrun.in/u/<handle>/w/project-text-analyser) works as the public URL too.