<# .SYNOPSIS Counts words per chapter file and for the whole manuscript, with an optional target. .DESCRIPTION Reads every .md, .txt and .docx file in a folder, in natural order (Chapter 2 before Chapter 10), and returns one object per file with its word count and a running total, then a Total row. Markdown is cleaned up before counting: YAML front matter, HTML comments (handy for notes to yourself), link targets, image tags and formatting marks are ignored, so **bold** and [links](url) count as the words you'd read. Word documents are read straight from the .docx package (it's a ZIP of XML), so Word doesn't need to be installed; tracked deletions, comments and footnotes aren't counted. A "word" is any run of non-space characters with at least one letter or digit in it, so a lone dash or *** scene break doesn't count. Give it a -Target and each row also shows the percentage reached, and you get a progress bar at the end. .PARAMETER Path A folder of chapter files, or specific files. Defaults to the current folder. .PARAMETER Include Extensions to count. Default: .md, .txt, .docx. .PARAMETER Recurse Include subfolders (for manuscripts split into part folders). .PARAMETER Target Your target word count for the whole manuscript. .PARAMETER ExcludeTotal Leave the Total row off, if you're feeding the output somewhere that sums it itself. .EXAMPLE .\Measure-ManuscriptWords.ps1 -Path .\chapters -Target 80000 .EXAMPLE .\Measure-ManuscriptWords.ps1 -Path .\chapters | Export-Csv wordcount.csv -NoTypeInformation #> [CmdletBinding()] param( [Parameter(ValueFromPipeline, ValueFromPipelineByPropertyName)] [Alias('FullName')] [string[]]$Path = '.', [string[]]$Include = @('.md', '.txt', '.docx'), [switch]$Recurse, [ValidateRange(1, 10000000)][int]$Target, [switch]$ExcludeTotal ) begin { Add-Type -AssemblyName System.IO.Compression, System.IO.Compression.FileSystem $Include = @($Include | ForEach-Object { if ($_ -like '.*') { $_.ToLower() } else { ".$_".ToLower() } }) $files = [System.Collections.Generic.List[IO.FileInfo]]::new() function Get-DocxText([string]$File) { $zip = [IO.Compression.ZipFile]::OpenRead($File) try { $entry = $zip.GetEntry('word/document.xml') if (-not $entry) { throw 'No word/document.xml inside. Is this really a .docx?' } $reader = [IO.StreamReader]::new($entry.Open()) try { [xml]$xml = $reader.ReadToEnd() } finally { $reader.Dispose() } } finally { $zip.Dispose() } $ns = [Xml.XmlNamespaceManager]::new($xml.NameTable) $ns.AddNamespace('w', 'http://schemas.openxmlformats.org/wordprocessingml/2006/main') # One line per paragraph; w:t holds the visible text, w:tab and w:br separate words. $paragraphs = foreach ($p in $xml.SelectNodes('//w:body//w:p', $ns)) { ($p.SelectNodes('.//w:t | .//w:tab | .//w:br', $ns) | ForEach-Object { if ($_.LocalName -eq 't') { $_.InnerText } else { ' ' } }) -join '' } $paragraphs -join "`n" } function Get-MarkdownText([string]$Text) { $Text = $Text -replace '\A', '' $Text = $Text -replace '(?s)\A---\r?\n.*?\r?\n(---|\.\.\.)\r?\n', '' # front matter $Text = $Text -replace '(?s)', '' # comments and notes to self $Text = $Text -replace '!\[[^\]]*\]\([^)]*\)', '' # images $Text = $Text -replace '\[([^\]]*)\]\([^)]*\)', '$1' # links: keep the text $Text = $Text -replace '<[^>]+>', ' ' # stray HTML tags $Text -replace '(?m)^\s{0,3}(#{1,6}|>|[-*+]|\d+\.)\s+', '' -replace '[*_~`]+', '' } function Measure-Word([string]$Text) { if (-not $Text) { return 0 } @($Text -split '\s+' | Where-Object { $_ -match '[\p{L}\p{N}]' }).Count } } process { foreach ($item in $Path) { if (Test-Path -LiteralPath $item -PathType Container) { Get-ChildItem -LiteralPath $item -File -Recurse:$Recurse | Where-Object { $Include -contains $_.Extension.ToLower() -and $_.Name -notlike '~$*' } | ForEach-Object { $files.Add($_) } } elseif (Test-Path -LiteralPath $item -PathType Leaf) { $files.Add((Get-Item -LiteralPath $item)) } else { Write-Warning "Not found: $item" } } } end { if (-not $files.Count) { Write-Warning 'No chapter files found.'; return } # Natural sort: pad every number so "Chapter 2" sorts before "Chapter 10". $sorted = $files | Sort-Object -Unique FullName | Sort-Object { $_.DirectoryName }, { [regex]::Replace($_.Name, '\d+', { $args[0].Value.PadLeft(10, '0') }) } $running = 0; $i = 0; $counted = 0 foreach ($file in $sorted) { $i++ Write-Progress -Activity 'Counting words' -Status $file.Name -PercentComplete (100 * $i / @($sorted).Count) $row = [ordered]@{ Chapter = $file.BaseName; File = $file.Name; Words = $null; RunningTotal = $null } try { $text = switch ($file.Extension.ToLower()) { '.docx' { Get-DocxText $file.FullName } '.md' { Get-MarkdownText (Get-Content -LiteralPath $file.FullName -Raw -Encoding UTF8 -ErrorAction Stop) } default { Get-Content -LiteralPath $file.FullName -Raw -Encoding UTF8 -ErrorAction Stop } } $row.Words = Measure-Word $text $running += $row.Words $counted++ } catch { Write-Warning "$($file.Name): $($_.Exception.Message)" } $row.RunningTotal = $running if ($Target) { $row.PercentOfTarget = [math]::Round(100 * $running / $Target, 1) } [pscustomobject]$row } Write-Progress -Activity 'Counting words' -Completed if (-not $ExcludeTotal) { $total = [ordered]@{ Chapter = 'Total'; File = "$counted file(s)"; Words = $running; RunningTotal = $running } if ($Target) { $total.PercentOfTarget = [math]::Round(100 * $running / $Target, 1) } [pscustomobject]$total } if ($Target) { $pct = [math]::Min(1.0, $running / $Target) $filled = [int][math]::Round(30 * $pct) $left = [math]::Max(0, $Target - $running) $bar = '[' + ('#' * $filled) + ('-' * (30 - $filled)) + ']' Write-Host ('{0} {1:N0} of {2:N0} words ({3:N0}%){4}' -f $bar, $running, $Target, (100 * $running / $Target), $(if ($left) { ", $($left.ToString('N0')) to go" } else { '. Done!' })) } }