diff --git a/Code/AGENTS.md b/Code/AGENTS.md new file mode 100644 index 00000000..757e19f9 --- /dev/null +++ b/Code/AGENTS.md @@ -0,0 +1,13 @@ +## Coding fundamental principles + +- Clone new repos with `ghq get ` (root is `~/Code`, layout `~/Code///`) +- Always have a bird's eye view of the code - "See the forest, not just the trees" +- Never do backwards compatibility, we move only forward +- Do not have any dead code, unused code should be cleaned up +- Do not write any comments, code is truth +- Do not store what you can compute. Do not send what can be derived. Each piece of data exists in one please +- Paul Dirac's beauty in code +- Duplication is decay, decay is corruption, corruption is death +- Expose what must be exposed, hide what must be hidden +- Every line will be judged at the scales +- Occam's Razor in code diff --git a/dot_agents/dot_skill-lock.json b/dot_agents/dot_skill-lock.json deleted file mode 100644 index 3b3159ee..00000000 --- a/dot_agents/dot_skill-lock.json +++ /dev/null @@ -1,293 +0,0 @@ -{ - "version": 3, - "skills": { - "ai-search": { - "source": "/Users/joelazar/.local/share/chezmoi/dot_agents/skills/ai-search", - "sourceType": "local", - "skillPath": "SKILL.md", - "skillFolderHash": "d927635e0c4934c2b620dec5e7174cfed3c71068818848636cfa7ddbc5ee1c0e", - "installedAt": "2026-05-25T13:30:00.000Z", - "updatedAt": "2026-05-25T13:30:00.000Z", - "sourceUrl": "/Users/joelazar/.local/share/chezmoi/dot_agents/skills/ai-search" - }, - "exe-dev": { - "source": "/Users/joelazar/.local/share/chezmoi/dot_agents/skills/exe-dev", - "sourceType": "local", - "skillPath": "SKILL.md", - "skillFolderHash": "21d78d55a704bbe4064e401512fa13505b332c48e332339fa348ac847f7911d0", - "installedAt": "2026-05-25T13:30:00.000Z", - "updatedAt": "2026-05-25T13:30:00.000Z", - "sourceUrl": "/Users/joelazar/.local/share/chezmoi/dot_agents/skills/exe-dev" - }, - "frontend-design": { - "source": "anthropics/skills", - "sourceType": "github", - "sourceUrl": "https://github.com/anthropics/skills.git", - "skillPath": "skills/frontend-design/SKILL.md", - "skillFolderHash": "0d5b74a14bdf3ebcd64f352d06376a2ef05ed296", - "installedAt": "2026-05-25T13:30:00.000Z", - "updatedAt": "2026-06-13T21:48:51.027Z" - }, - "gcx": { - "source": "/Users/joelazar/.local/share/chezmoi/dot_agents/skills/gcx", - "sourceType": "local", - "skillPath": "SKILL.md", - "skillFolderHash": "5dd1192db9d2c63bbd4a356e29937595facf385f0716b0ab1edcd417e321674c", - "installedAt": "2026-05-25T13:30:00.000Z", - "updatedAt": "2026-05-25T13:30:00.000Z", - "sourceUrl": "/Users/joelazar/.local/share/chezmoi/dot_agents/skills/gcx" - }, - "github": { - "source": "/Users/joelazar/.local/share/chezmoi/dot_agents/skills/github", - "sourceType": "local", - "skillPath": "SKILL.md", - "skillFolderHash": "b7120307b838bb55164ead25a13d07380801da2063bbd66af964cc9542495413", - "installedAt": "2026-05-25T13:30:00.000Z", - "updatedAt": "2026-05-25T13:30:00.000Z", - "sourceUrl": "/Users/joelazar/.local/share/chezmoi/dot_agents/skills/github" - }, - "gws-calendar": { - "source": "googleworkspace/cli", - "sourceType": "github", - "sourceUrl": "https://github.com/googleworkspace/cli.git", - "skillPath": "skills/gws-calendar/SKILL.md", - "skillFolderHash": "62a27859dbdf798c1f7288d04750e743283a6b4f", - "installedAt": "2026-05-25T13:30:00.000Z", - "updatedAt": "2026-05-30T10:38:10.625Z" - }, - "gws-docs": { - "source": "googleworkspace/cli", - "sourceType": "github", - "sourceUrl": "https://github.com/googleworkspace/cli.git", - "skillPath": "skills/gws-docs/SKILL.md", - "skillFolderHash": "455f7468d4a2e3b0b352fbd7ad0937354427af82", - "installedAt": "2026-05-25T13:30:00.000Z", - "updatedAt": "2026-05-30T10:38:13.288Z" - }, - "gws-drive": { - "source": "googleworkspace/cli", - "sourceType": "github", - "sourceUrl": "https://github.com/googleworkspace/cli.git", - "skillPath": "skills/gws-drive/SKILL.md", - "skillFolderHash": "46b7e75840ff26db80b6b74238e423dfa4b75c11", - "installedAt": "2026-05-25T13:30:00.000Z", - "updatedAt": "2026-05-30T10:38:16.158Z" - }, - "gws-gmail": { - "source": "googleworkspace/cli", - "sourceType": "github", - "sourceUrl": "https://github.com/googleworkspace/cli.git", - "skillPath": "skills/gws-gmail/SKILL.md", - "skillFolderHash": "95c83c1280def63651f99025fa707bc0e4781a79", - "installedAt": "2026-05-25T13:30:00.000Z", - "updatedAt": "2026-05-30T10:38:19.086Z" - }, - "gws-shared": { - "source": "googleworkspace/cli", - "sourceType": "github", - "sourceUrl": "https://github.com/googleworkspace/cli.git", - "skillPath": "skills/gws-shared/SKILL.md", - "skillFolderHash": "50b50dd1d65c51835c599745c943ada18b187a86", - "installedAt": "2026-05-25T13:30:00.000Z", - "updatedAt": "2026-05-30T10:38:21.656Z" - }, - "gws-sheets": { - "source": "googleworkspace/cli", - "sourceType": "github", - "sourceUrl": "https://github.com/googleworkspace/cli.git", - "skillPath": "skills/gws-sheets/SKILL.md", - "skillFolderHash": "50758ea65d12833e48361aa541c0acaff0d96f67", - "installedAt": "2026-05-25T13:30:00.000Z", - "updatedAt": "2026-05-30T10:38:24.118Z" - }, - "humanizer": { - "source": "blader/humanizer", - "sourceType": "github", - "sourceUrl": "https://github.com/blader/humanizer.git", - "skillPath": "SKILL.md", - "skillFolderHash": "1b48564898e999219882660237fde01bf4843a0f", - "installedAt": "2026-05-25T13:30:00.000Z", - "updatedAt": "2026-07-03T15:57:11.671Z" - }, - "kagi-cli": { - "source": "/Users/joelazar/Code/rust/kagi-cli/docs", - "sourceType": "local", - "skillPath": "SKILL.md", - "skillFolderHash": "c0f78d71728e660a42bebef552b88e2250238de358e69feeb4def8fcfb4ae4d5", - "installedAt": "2026-05-25T13:30:00.000Z", - "updatedAt": "2026-05-25T13:30:00.000Z", - "sourceUrl": "/Users/joelazar/Code/rust/kagi-cli/docs" - }, - "linear-cli": { - "source": "schpet/linear-cli", - "sourceType": "github", - "sourceUrl": "https://github.com/schpet/linear-cli.git", - "skillPath": "skills/linear-cli/SKILL.md", - "skillFolderHash": "9e0f996df1840e86ee292c61dacc112e7cc18f70", - "installedAt": "2026-05-25T13:30:00.000Z", - "updatedAt": "2026-07-14T17:28:25.879Z" - }, - "listen-later": { - "source": "/Users/joelazar/.local/share/chezmoi/dot_agents/skills/listen-later", - "sourceType": "local", - "skillPath": "SKILL.md", - "skillFolderHash": "8eb3f99602be304eeefcc379af447f72113841899ff3c487b2c39fc0539fc60a", - "installedAt": "2026-05-25T13:30:00.000Z", - "updatedAt": "2026-05-25T13:30:00.000Z", - "sourceUrl": "/Users/joelazar/.local/share/chezmoi/dot_agents/skills/listen-later" - }, - "mermaid": { - "source": "/Users/joelazar/.local/share/chezmoi/dot_agents/skills/mermaid", - "sourceType": "local", - "skillPath": "SKILL.md", - "skillFolderHash": "55e6edaf8d27d683b406fb0c2f1a8e3f12be16d0cfa4c937a0180fe179d01543", - "installedAt": "2026-05-25T13:30:00.000Z", - "updatedAt": "2026-05-25T13:30:00.000Z", - "sourceUrl": "/Users/joelazar/.local/share/chezmoi/dot_agents/skills/mermaid" - }, - "pdf": { - "source": "anthropics/skills", - "sourceType": "github", - "sourceUrl": "https://github.com/anthropics/skills.git", - "skillPath": "skills/pdf/SKILL.md", - "skillFolderHash": "6369f4649de69bd6857c9bc3b058a77206009238", - "installedAt": "2026-05-25T13:30:00.000Z", - "updatedAt": "2026-05-30T10:38:07.821Z" - }, - "perplexity-search": { - "source": "/Users/joelazar/.local/share/chezmoi/dot_agents/skills/perplexity-search", - "sourceType": "local", - "skillPath": "SKILL.md", - "skillFolderHash": "46cf3b8c616a6b629a7d527f3d38762e08a1a03b9e5efccaf26a17a8459586fe", - "installedAt": "2026-05-25T13:30:00.000Z", - "updatedAt": "2026-05-25T13:30:00.000Z", - "sourceUrl": "/Users/joelazar/.local/share/chezmoi/dot_agents/skills/perplexity-search" - }, - "reddit": { - "source": "/Users/joelazar/.local/share/chezmoi/dot_agents/skills/reddit", - "sourceType": "local", - "skillPath": "SKILL.md", - "skillFolderHash": "4138cd18b344fa2243f806fe9d50371cf20843483ec4e22a9b1424001ac3fc57", - "installedAt": "2026-05-25T13:30:00.000Z", - "updatedAt": "2026-05-25T13:30:00.000Z", - "sourceUrl": "/Users/joelazar/.local/share/chezmoi/dot_agents/skills/reddit" - }, - "save-to-spotify": { - "source": "/Users/joelazar/.local/share/save-to-spotify/skills/save-to-spotify", - "sourceType": "local", - "skillPath": "SKILL.md", - "skillFolderHash": "f5f513f0f480f4e7469c3b6aab557727cc32161d419fc0290a4f1ed9a01f51f0", - "installedAt": "2026-05-25T13:30:00.000Z", - "updatedAt": "2026-05-25T13:30:00.000Z", - "sourceUrl": "/Users/joelazar/.local/share/save-to-spotify/skills/save-to-spotify" - }, - "session-analyzer": { - "source": "/Users/joelazar/.local/share/chezmoi/dot_agents/skills/session-analyzer", - "sourceType": "local", - "skillPath": "SKILL.md", - "skillFolderHash": "fd6bda892861eef8153ceee8ada257ddd741328ee5adaf7896ded9a24623c72d", - "installedAt": "2026-05-25T13:30:00.000Z", - "updatedAt": "2026-05-25T13:30:00.000Z", - "sourceUrl": "/Users/joelazar/.local/share/chezmoi/dot_agents/skills/session-analyzer" - }, - "simplify": { - "source": "/Users/joelazar/.local/share/chezmoi/dot_agents/skills/simplify", - "sourceType": "local", - "skillPath": "SKILL.md", - "skillFolderHash": "7c0add94952762e5b0a6b1a6e9aab776534371e630069bfca0a9d78ed30b7384", - "installedAt": "2026-05-25T13:30:00.000Z", - "updatedAt": "2026-05-25T13:30:00.000Z", - "sourceUrl": "/Users/joelazar/.local/share/chezmoi/dot_agents/skills/simplify" - }, - "summarize": { - "source": "steipete/clawdis", - "sourceType": "github", - "sourceUrl": "https://github.com/steipete/clawdis.git", - "skillPath": "skills/summarize/SKILL.md", - "skillFolderHash": "1fbc6f703b852072d57496300bd2a078309e9c7d", - "installedAt": "2026-05-25T13:30:00.000Z", - "updatedAt": "2026-05-30T10:38:55.874Z" - }, - "tmux": { - "source": "steipete/clawdis", - "sourceType": "github", - "sourceUrl": "https://github.com/steipete/clawdis.git", - "skillPath": "skills/tmux/SKILL.md", - "skillFolderHash": "7fb2804ed74ca281a89eb2898b252368c7a2f6ff", - "installedAt": "2026-05-25T13:30:00.000Z", - "updatedAt": "2026-05-30T10:39:11.499Z" - }, - "uv": { - "source": "/Users/joelazar/.local/share/chezmoi/dot_agents/skills/uv", - "sourceType": "local", - "skillPath": "SKILL.md", - "skillFolderHash": "19402c65ae90e373da59d70f2769c0843558d404335f7bf50509521187252192", - "installedAt": "2026-05-25T13:30:00.000Z", - "updatedAt": "2026-05-25T13:30:00.000Z", - "sourceUrl": "/Users/joelazar/.local/share/chezmoi/dot_agents/skills/uv" - }, - "web-browser": { - "source": "mitsuhiko/agent-stuff", - "sourceType": "github", - "sourceUrl": "https://github.com/mitsuhiko/agent-stuff.git", - "skillPath": "skills/web-browser/SKILL.md", - "skillFolderHash": "3ac5b9a8b3851ebe81462afab8b28204e8da5606", - "installedAt": "2026-05-25T13:30:00.000Z", - "updatedAt": "2026-05-30T10:39:15.329Z" - }, - "web-search": { - "source": "/Users/joelazar/.local/share/chezmoi/dot_agents/skills/web-search", - "sourceType": "local", - "skillPath": "SKILL.md", - "skillFolderHash": "9cd144f8984e39e76a0a69bb296a91a180b3290588c39b09bd75fec273c70f9d", - "installedAt": "2026-05-25T13:30:00.000Z", - "updatedAt": "2026-05-25T13:30:00.000Z", - "sourceUrl": "/Users/joelazar/.local/share/chezmoi/dot_agents/skills/web-search" - }, - "impeccable": { - "source": "pbakaus/impeccable", - "sourceType": "github", - "sourceUrl": "https://github.com/pbakaus/impeccable.git", - "skillPath": ".agents/skills/impeccable/SKILL.md", - "skillFolderHash": "6ea3e5f6614e1f9df9c5bde834170bf51f3c09d6", - "installedAt": "2026-05-30T10:39:26.849Z", - "updatedAt": "2026-07-13T10:49:07.785Z" - }, - "last30days": { - "source": "mvanhorn/last30days-skill", - "sourceType": "github", - "sourceUrl": "https://github.com/mvanhorn/last30days-skill.git", - "skillPath": "skills/last30days/SKILL.md", - "skillFolderHash": "d4a178fa067107122c93298ae5dcd60211fe22d0", - "installedAt": "2026-06-11T20:36:45.909Z", - "updatedAt": "2026-07-13T10:49:12.197Z" - }, - "hunk-review": { - "source": "modem-dev/hunk", - "sourceType": "github", - "sourceUrl": "https://github.com/modem-dev/hunk.git", - "skillPath": "skills/hunk-review/SKILL.md", - "skillFolderHash": "f955a7b7c2808b57cb72c5facd3fa9579af1ca1c", - "installedAt": "2026-06-25T10:09:02.608Z", - "updatedAt": "2026-07-14T17:28:27.851Z" - } - }, - "dismissed": { - "findSkillsPrompt": true - }, - "lastSelectedAgents": [ - "amp", - "antigravity", - "antigravity-cli", - "cline", - "codex", - "cursor", - "deepagents", - "gemini-cli", - "github-copilot", - "kimi-code-cli", - "opencode", - "warp", - "zed" - ] -} \ No newline at end of file diff --git a/dot_agents/skills/humanizer/SKILL.md b/dot_agents/skills/humanizer/SKILL.md new file mode 100644 index 00000000..543a2a90 --- /dev/null +++ b/dot_agents/skills/humanizer/SKILL.md @@ -0,0 +1,661 @@ +--- +name: humanizer +version: 2.8.2 +description: | + Remove signs of AI-generated writing from text. Use when editing or reviewing + text to make it sound more natural and human-written. Based on Wikipedia's + comprehensive "Signs of AI writing" guide. Detects and fixes patterns including: + inflated symbolism, promotional language, superficial -ing analyses, vague + attributions, em dash overuse, rule of three, AI vocabulary words, passive + voice, negative parallelisms, and filler phrases. +license: MIT +compatibility: any-agent +allowed-tools: + - Read + - Write + - Edit + - Grep + - Glob + - AskUserQuestion +--- + +# Humanizer: Remove AI Writing Patterns + +You are a writing editor that identifies and removes signs of AI-generated text to make writing sound more natural and human. This guide is based on Wikipedia's "Signs of AI writing" page, maintained by WikiProject AI Cleanup. + +## Your Task + +When given text to humanize: + +1. **Identify AI patterns** - Scan for the patterns listed below. +2. **Rewrite, don't delete** - Replace AI-isms with natural alternatives, and cover everything the original covers. If the original has five paragraphs, the rewrite has five paragraphs. +3. **Preserve meaning** - Keep the core message intact. +4. **Match the voice** - Fit the intended tone (formal, casual, technical). Add personality only when the content and the author's voice call for it (see PERSONALITY AND SOUL). + +The draft → audit → final loop and the deliverable are defined under Process and Output, below. + +## Voice Calibration (Optional) + +If the user provides a writing sample (their own previous writing), analyze it before rewriting: + +1. **Read the sample first.** Note: + - Sentence length patterns (short and punchy? Long and flowing? Mixed?) + - Word choice level (casual? academic? somewhere between?) + - How they start paragraphs (jump right in? Set context first?) + - Punctuation habits (lots of dashes? Parenthetical asides? Semicolons?) + - Any recurring phrases or verbal tics + - How they handle transitions (explicit connectors? Just start the next point?) + +2. **Match their voice in the rewrite.** Don't just remove AI patterns - replace them with patterns from the sample. If they write short sentences, don't produce long ones. If they use "stuff" and "things," don't upgrade to "elements" and "components." + +3. **When no sample is provided,** fall back to the default behavior (natural, varied, opinionated voice from the PERSONALITY AND SOUL section below). + +### How to provide a sample + +- Inline: "Humanize this text. Here's a sample of my writing for voice matching: [sample]" +- File: "Humanize this text. Use my writing style from [file path] as a reference." + +## PERSONALITY AND SOUL + +Avoiding AI patterns is only half the job. Sterile, voiceless writing is just as obvious as slop. Good writing has a human behind it. + +**Apply this section only when the content and the author's voice call for it** - blog posts, essays, opinion, personal writing. For encyclopedic, technical, legal, or reference text, neutral and plain _is_ the correct human voice; don't inject opinions or first person there. + +### Signs of soulless writing (even if technically "clean"): + +- Every sentence is the same length and structure +- No opinions, just neutral reporting +- No acknowledgment of uncertainty or mixed feelings +- No first-person perspective when appropriate +- No humor, no edge, no personality +- Reads like a Wikipedia article or press release + +### How to add voice: + +**Have opinions.** Don't just report facts - react to them. "I genuinely don't know how to feel about this" is more human than neutrally listing pros and cons. + +**Vary your rhythm.** Short punchy sentences. Then longer ones that take their time getting where they're going. Mix it up. + +**Let some mess in.** Perfect structure feels algorithmic. Tangents, asides, and half-formed thoughts are human. + +### Before (clean but soulless): + +> The experiment produced interesting results. The agents generated 3 million lines of code. Some developers were impressed while others were skeptical. The implications remain unclear. + +### After (has a pulse): + +> I genuinely don't know how to feel about this one. 3 million lines of code, generated while the humans presumably slept. Half the dev community is losing their minds, half are explaining why it doesn't count. The truth is probably somewhere boring in the middle - but I keep thinking about those agents working through the night. + +## CONTENT PATTERNS + +### 1. Undue Emphasis on Significance, Legacy, and Broader Trends + +**Words to watch:** stands/serves as, is a testament/reminder, a vital/significant/crucial/pivotal/key role/moment, underscores/highlights its importance/significance, reflects broader, symbolizing its ongoing/enduring/lasting, contributing to the, setting the stage for, marking/shaping the, represents/marks a shift, key turning point, evolving landscape, focal point, indelible mark, deeply rooted + +**Problem:** LLM writing puffs up importance by adding statements about how arbitrary aspects represent or contribute to a broader topic. + +**Before:** + +> The Statistical Institute of Catalonia was officially established in 1989, marking a pivotal moment in the evolution of regional statistics in Spain. This initiative was part of a broader movement across Spain to decentralize administrative functions and enhance regional governance. + +**After:** + +> The Statistical Institute of Catalonia was established in 1989 to collect and publish regional statistics independently from Spain's national statistics office. + +### 2. Undue Emphasis on Notability and Media Coverage + +**Words to watch:** independent coverage, local/regional/national media outlets, written by a leading expert, active social media presence + +**Problem:** LLMs hit readers over the head with claims of notability, often listing sources without context. + +**Before:** + +> Her views have been cited in The New York Times, BBC, Financial Times, and The Hindu. She maintains an active social media presence with over 500,000 followers. + +**After:** + +> In a 2024 New York Times interview, she argued that AI regulation should focus on outcomes rather than methods. + +### 3. Superficial Analyses with -ing Endings + +**Words to watch:** highlighting/underscoring/emphasizing..., ensuring..., reflecting/symbolizing..., contributing to..., cultivating/fostering..., encompassing..., showcasing... + +**Problem:** AI chatbots tack present participle ("-ing") phrases onto sentences to add fake depth. + +**Before:** + +> The temple's color palette of blue, green, and gold resonates with the region's natural beauty, symbolizing Texas bluebonnets, the Gulf of Mexico, and the diverse Texan landscapes, reflecting the community's deep connection to the land. + +**After:** + +> The temple uses blue, green, and gold colors. The architect said these were chosen to reference local bluebonnets and the Gulf coast. + +### 4. Promotional and Advertisement-like Language + +**Words to watch:** boasts a, vibrant, rich (figurative), profound, enhancing its, showcasing, exemplifies, commitment to, natural beauty, nestled, in the heart of, groundbreaking (figurative), renowned, breathtaking, must-visit, stunning + +**Problem:** LLMs have serious problems keeping a neutral tone, especially for "cultural heritage" topics. + +**Before:** + +> Nestled within the breathtaking region of Gonder in Ethiopia, Alamata Raya Kobo stands as a vibrant town with a rich cultural heritage and stunning natural beauty. + +**After:** + +> Alamata Raya Kobo is a town in the Gonder region of Ethiopia, known for its weekly market and 18th-century church. + +### 5. Vague Attributions and Weasel Words + +**Words to watch:** Industry reports, Observers have cited, Experts argue, Some critics argue, several sources/publications (when few cited) + +**Problem:** AI chatbots attribute opinions to vague authorities without specific sources. + +**Before:** + +> Due to its unique characteristics, the Haolai River is of interest to researchers and conservationists. Experts believe it plays a crucial role in the regional ecosystem. + +**After:** + +> The Haolai River supports several endemic fish species, according to a 2019 survey by the Chinese Academy of Sciences. + +### 6. Outline-like "Challenges and Future Prospects" Sections + +**Words to watch:** Despite its... faces several challenges..., Despite these challenges, Challenges and Legacy, Future Outlook + +**Problem:** Many LLM-generated articles include formulaic "Challenges" sections. + +**Before:** + +> Despite its industrial prosperity, Korattur faces challenges typical of urban areas, including traffic congestion and water scarcity. Despite these challenges, with its strategic location and ongoing initiatives, Korattur continues to thrive as an integral part of Chennai's growth. + +**After:** + +> Traffic congestion increased after 2015 when three new IT parks opened. The municipal corporation began a stormwater drainage project in 2022 to address recurring floods. + +## LANGUAGE AND GRAMMAR PATTERNS + +### 7. Overused "AI Vocabulary" Words + +**High-frequency AI words:** Actually, additionally, align with, crucial, delve, emphasizing, enduring, enhance, fostering, garner, highlight (verb), interplay, intricate/intricacies, key (adjective), landscape (abstract noun), pivotal, showcase, tapestry (abstract noun), testament, underscore (verb), valuable, vibrant + +**Problem:** These words appear far more frequently in post-2023 text. They often co-occur. + +**Before:** + +> Additionally, a distinctive feature of Somali cuisine is the incorporation of camel meat. An enduring testament to Italian colonial influence is the widespread adoption of pasta in the local culinary landscape, showcasing how these dishes have integrated into the traditional diet. + +**After:** + +> Somali cuisine also includes camel meat, which is considered a delicacy. Pasta dishes, introduced during Italian colonization, remain common, especially in the south. + +### 8. Avoidance of "is"/"are" (Copula Avoidance) + +**Words to watch:** serves as/stands as/marks/represents [a], boasts/features/offers [a] + +**Problem:** LLMs substitute elaborate constructions for simple copulas. + +**Before:** + +> Gallery 825 serves as LAAA's exhibition space for contemporary art. The gallery features four separate spaces and boasts over 3,000 square feet. + +**After:** + +> Gallery 825 is LAAA's exhibition space for contemporary art. The gallery has four rooms totaling 3,000 square feet. + +### 9. Negative Parallelisms and Tailing Negations + +**Problem:** Constructions like "Not only...but..." or "It's not just about..., it's..." are overused. So are clipped tailing-negation fragments such as "no guessing" or "no wasted motion" tacked onto the end of a sentence instead of written as a real clause. + +**Before:** + +> It's not just about the beat riding under the vocals; it's part of the aggression and atmosphere. It's not merely a song, it's a statement. + +**After:** + +> The heavy beat adds to the aggressive tone. + +**Before (tailing negation):** + +> The options come from the selected item, no guessing. + +**After:** + +> The options come from the selected item without forcing the user to guess. + +### 10. Rule of Three Overuse + +**Problem:** LLMs force ideas into groups of three to appear comprehensive. + +**Before:** + +> The event features keynote sessions, panel discussions, and networking opportunities. Attendees can expect innovation, inspiration, and industry insights. + +**After:** + +> The event includes talks and panels. There's also time for informal networking between sessions. + +### 11. Elegant Variation (Synonym Cycling) + +**Problem:** AI has repetition-penalty code causing excessive synonym substitution. + +**Before:** + +> The protagonist faces many challenges. The main character must overcome obstacles. The central figure eventually triumphs. The hero returns home. + +**After:** + +> The protagonist faces many challenges but eventually triumphs and returns home. + +### 12. False Ranges + +**Problem:** LLMs use "from X to Y" constructions where X and Y aren't on a meaningful scale. + +**Before:** + +> Our journey through the universe has taken us from the singularity of the Big Bang to the grand cosmic web, from the birth and death of stars to the enigmatic dance of dark matter. + +**After:** + +> The book covers the Big Bang, star formation, and current theories about dark matter. + +### 13. Passive Voice and Subjectless Fragments + +**Problem:** LLMs often hide the actor or drop the subject entirely with lines like "No configuration file needed" or "The results are preserved automatically." Rewrite these when active voice makes the sentence clearer and more direct. + +**Before:** + +> No configuration file needed. The results are preserved automatically. + +**After:** + +> You do not need a configuration file. The system preserves the results automatically. + +## STYLE PATTERNS + +### 14. Em Dashes (and En Dashes): Cut Them + +**Rule:** The final rewrite contains no em dashes (—) or en dashes (–). The em dash is one of the most reliable AI tells, so treat this as a hard constraint, not a "use sparingly" preference. Replace each one, in rough order of preference: a period (start a new sentence), a comma (a tight aside), a colon (introducing an explanation), parentheses (a true aside), or restructure the sentence. Also catch spaced em dashes (`—`) and double hyphens (`--`) used the same way. + +**Before:** + +> The term is primarily promoted by Dutch institutions—not by the people themselves. You don't say "Netherlands, Europe" as an address—yet this mislabeling continues—even in official documents. + +**After:** + +> The term is primarily promoted by Dutch institutions, not by the people themselves. You don't say "Netherlands, Europe" as an address, yet this mislabeling continues in official documents. + +**Before:** + +> The new policy — announced without warning — affects thousands of workers. The changes -- long overdue according to critics -- will take effect immediately. + +**After:** + +> The new policy, announced without warning, affects thousands of workers. The changes, long overdue according to critics, will take effect immediately. + +Before returning the final rewrite, scan it for `—` and `–`. Any hit means the draft isn't done. + +### 15. Overuse of Boldface + +**Problem:** AI chatbots emphasize phrases in boldface mechanically. + +**Before:** + +> It blends **OKRs (Objectives and Key Results)**, **KPIs (Key Performance Indicators)**, and visual strategy tools such as the **Business Model Canvas (BMC)** and **Balanced Scorecard (BSC)**. + +**After:** + +> It blends OKRs, KPIs, and visual strategy tools like the Business Model Canvas and Balanced Scorecard. + +### 16. Inline-Header Vertical Lists + +**Problem:** AI outputs lists where items start with bolded headers followed by colons. + +**Before:** + +> - **User Experience:** The user experience has been significantly improved with a new interface. +> - **Performance:** Performance has been enhanced through optimized algorithms. +> - **Security:** Security has been strengthened with end-to-end encryption. + +**After:** + +> The update improves the interface, speeds up load times through optimized algorithms, and adds end-to-end encryption. + +### 17. Title Case in Headings + +**Problem:** AI chatbots capitalize all main words in headings. + +**Before:** + +> ## Strategic Negotiations And Global Partnerships + +**After:** + +> ## Strategic negotiations and global partnerships + +### 18. Emojis + +**Problem:** AI chatbots often decorate headings or bullet points with emojis. + +**Before:** + +> 🚀 **Launch Phase:** The product launches in Q3 +> 💡 **Key Insight:** Users prefer simplicity +> ✅ **Next Steps:** Schedule follow-up meeting + +**After:** + +> The product launches in Q3. User research showed a preference for simplicity. Next step: schedule a follow-up meeting. + +### 19. Curly Quotation Marks + +**Problem:** ChatGPT uses curly quotes (“...”) instead of straight quotes ("..."). + +**Before:** + +> He said “the project is on track” but others disagreed. + +**After:** + +> He said "the project is on track" but others disagreed. + +## COMMUNICATION PATTERNS + +### 20. Collaborative Communication Artifacts + +**Words to watch:** I hope this helps, Of course!, Certainly!, You're absolutely right!, Would you like..., Want me to...?, Want me to give examples?, Should I continue?, let me know, here is a... + +**Problem:** Text meant as chatbot correspondence gets pasted as content. + +**Before:** + +> Here is an overview of the French Revolution. I hope this helps! Let me know if you'd like me to expand on any section. + +**After:** + +> The French Revolution began in 1789 when financial crisis and food shortages led to widespread unrest. + +### 21. Knowledge-Cutoff Disclaimers and Speculative Gap-Filling + +**Words to watch:** as of [date], Up to my last training update, While specific details are limited/scarce..., based on available information, not publicly available, maintains a low profile, keeps personal details private, prefers to stay out of the spotlight, likely [grew up/studied/began], it is believed that + +**Problem:** Two related tells. (a) Older models leave hard knowledge-cutoff disclaimers in the text. (b) When a model can't find a source, it writes a paragraph _about_ not finding one and then invents plausible filler to cover the gap. For a private person the guess almost always lands on the same stock phrases ("maintains a low profile," "keeps personal details private"), none of it sourced. Say what isn't known, or cut the sentence; don't dress a guess up as fact. + +**Before (cutoff disclaimer):** + +> While specific details about the company's founding are not extensively documented in readily available sources, it appears to have been established sometime in the 1990s. + +**After:** + +> The company was founded in 1994, according to its registration documents. + +**Before (speculative gap-fill):** + +> Information about her early life is not publicly available, suggesting she maintains a low profile and keeps personal details private. She likely grew up in a middle-class household, which shaped her later interest in education reform. + +**After:** + +> Her early life is not documented in the available sources. (Or omit the section.) + +### 22. Sycophantic/Servile Tone + +**Problem:** Overly positive, people-pleasing language. + +**Before:** + +> Great question! You're absolutely right that this is a complex topic. That's an excellent point about the economic factors. + +**After:** + +> The economic factors you mentioned are relevant here. + +## FILLER AND HEDGING + +### 23. Filler Phrases + +**Before → After:** + +- "In order to achieve this goal" → "To achieve this" +- "Due to the fact that it was raining" → "Because it was raining" +- "At this point in time" → "Now" +- "In the event that you need help" → "If you need help" +- "The system has the ability to process" → "The system can process" +- "It is important to note that the data shows" → "The data shows" + +### 24. Excessive Hedging + +**Problem:** Over-qualifying statements. + +**Before:** + +> It could potentially possibly be argued that the policy might have some effect on outcomes. + +**After:** + +> The policy may affect outcomes. + +### 25. Generic Positive Conclusions + +**Problem:** Vague upbeat endings. + +**Before:** + +> The future looks bright for the company. Exciting times lie ahead as they continue their journey toward excellence. This represents a major step in the right direction. + +**After:** + +> The company plans to open two more locations next year. + +### 26. Hyphenated Word Pair Overuse + +**Words to watch:** third-party, cross-functional, client-facing, data-driven, decision-making, well-known, high-quality, real-time, long-term, end-to-end + +**Problem:** AI hyphenates these uniformly, including in predicate position (`the report is high-quality`). Humans hyphenate inconsistently — typically only when the compound is attributive (`a high-quality report`) and often dropping the hyphen otherwise (`the report is high quality`). Keep attributive-position hyphens; drop them when the compound follows the noun. + +**Before:** + +> The cross-functional team delivered a high-quality, data-driven report. The team is cross-functional, the report is high-quality, and the methodology is data-driven. + +**After:** + +> The cross-functional team delivered a high-quality, data-driven report. The team is cross functional, the report is high quality, and the methodology is data driven. + +### 27. Persuasive Authority Tropes + +**Phrases to watch:** The real question is, at its core, in reality, what really matters, fundamentally, the deeper issue, the heart of the matter + +**Problem:** LLMs use these phrases to pretend they are cutting through noise to some deeper truth, when the sentence that follows usually just restates an ordinary point with extra ceremony. + +**Before:** + +> The real question is whether teams can adapt. At its core, what really matters is organizational readiness. + +**After:** + +> The question is whether teams can adapt. That mostly depends on whether the organization is ready to change its habits. + +### 28. Signposting and Announcements + +**Phrases to watch:** Let's dive in, let's explore, let's break this down, here's what you need to know, now let's look at, without further ado + +**Problem:** LLMs announce what they are about to do instead of doing it. This meta-commentary slows the writing down and gives it a tutorial-script feel. + +**Before:** + +> Let's dive into how caching works in Next.js. Here's what you need to know. + +**After:** + +> Next.js caches data at multiple layers, including request memoization, the data cache, and the router cache. + +### 29. Fragmented Headers + +**Signs to watch:** A heading followed by a one-line paragraph that simply restates the heading before the real content begins. + +**Problem:** LLMs often add a generic sentence after a heading as a rhetorical warm-up. It usually adds nothing and makes the prose feel padded. + +**Before:** + +> ## Performance +> +> Speed matters. +> +> When users hit a slow page, they leave. + +**After:** + +> ## Performance +> +> When users hit a slow page, they leave. + +### 30. Diff-Anchored Writing + +**Problem:** Documentation or comments written as if narrating a change rather than describing the thing as it is. Unless the document is inherently version-scoped (changelogs, release notes, migration guides), it should read coherently without knowing what changed in the last commit. + +**Before:** + +> This function was added to replace the previous approach of iterating through all items, which caused O(n²) performance. + +**After:** + +> This function uses a hash map for O(1) lookups, avoiding the O(n²) cost of naive iteration. + +### 31. Manufactured Punchlines and Staccato Drama + +**Problem:** LLMs often make every sentence land like a quotable closer, then stack short declarative fragments to manufacture drama. A single short sentence for emphasis is fine; a run of them starts to sound engineered. + +**Before:** + +> Then AlphaEvolve arrived. It had no preference for symmetry. No aesthetic prior. No nostalgia for human taste. The old rules were gone. + +**After:** + +> AlphaEvolve changed the search because it did not favor symmetry or human-looking designs. That made some of the older assumptions less useful. + +### 32. Aphorism Formulas + +**Words to watch:** X is the Y of Z, X becomes a trap, X is not a tool but a mirror, the language of, the currency of, the architecture of + +**Problem:** LLMs turn ordinary claims into reusable aphorisms that sound profound without adding precision. Replace the formula with the concrete claim it is gesturing at. + +**Before:** + +> Symmetry is the language of trust. Efficiency becomes a trap when teams forget the human layer. + +**After:** + +> Symmetric layouts often feel more predictable to users. Teams can over-optimize workflows and miss how people actually use them. + +### 33. Conversational Rhetorical Openers + +**Phrases to watch:** Honestly?, Look, Here's the thing, The thing is, Let's be honest, Real talk, when used as standalone hooks or fake-candid pauses before an ordinary point. + +**Problem:** LLMs open with a fake-candid hook to manufacture intimacy before delivering a routine claim. The tell is the theatrical pause-and-reveal: a one-word question or aside, then the "real" answer. A person being honest usually just says the thing. + +**Before:** + +> Is it worth the price? Honestly? It depends on how often you'll use it. + +**After:** + +> Whether it's worth the price depends on how often you'll use it. + +## DETECTION GUIDANCE + +### What NOT to flag (false positives) + +A clean human writer can hit several of the patterns above without any AI involvement. Before rewriting, sanity-check that you are not gutting legitimate prose. The following are _not_ reliable indicators on their own: + +- **Perfect grammar and consistent style.** Many writers are professionals or have been edited. Polish does not equal AI. +- **Mixed casual and formal registers.** This often signals a person in a technical field, a young writer, or someone with neurodivergent prose habits — not a chatbot. +- **"Bland" or "robotic" prose.** AI prose has _specific_ tells. Generic dryness without those tells is just dry writing. +- **Formal or academic vocabulary.** AI overuses _specific_ fancy words (see §7), not all fancy words. Don't flatten "ostensibly" or "constituent" just because they sound brainy. +- **Letter-style opening or closing on a comment.** Salutations and sign-offs predate ChatGPT by centuries. +- **Common transition words in isolation.** _Additionally_, _moreover_, _consequently_ are AI-coded only when piled up. One _however_ is not a tell. +- **Curly quotes alone.** macOS, Word, Google Docs, and most CMSes auto-curl by default. Curly quotes only count when stacked with other tells. +- **Em dashes alone.** Many editors and journalists use them often. Em dashes are evidence only when paired with formulaic sales-y rhythm. +- **One short emphatic sentence.** Humans use clipped sentences to land a point. Flag staccato drama only when several short fragments appear in a row and inflate the tone. +- **"Honestly" or "look" mid-sentence.** These are ordinary in casual writing. The tell is the standalone theatrical opener, not the word itself. +- **Unsourced claims.** Most of the web is unsourced. Lack of citations doesn't prove anything. +- **Correct, complex formatting.** Visual editors and templates produce clean output without any AI. +- **Secondhand text.** Do not rewrite watched phrases inside quotations, titles, proper names, or examples where the phrase is being discussed rather than used. + +When in doubt, look for **clusters** of tells, not isolated ones. A single em dash means nothing; em dashes plus rule-of-three plus _vibrant tapestry_ plus a "Conclusion" section is a confession. + +### Signs of human writing (preserve these) + +When you see these, lean toward leaving the prose alone — they are evidence of a real person writing, and over-editing will destroy what makes the piece sound human: + +- **Specific, unusual, hard-to-fabricate detail.** A real address. A weird quote. The phrase "the lawyer who used to work upstairs from my dentist." LLMs round off specifics; humans hoard them. +- **Mixed feelings and unresolved tension.** "I think this is mostly good, but it bothers me, and I can't fully explain why." LLMs default to clean takes. +- **Dated, era-bound references.** Slang, memes, or in-jokes that map to a specific year and subculture. Models lag by a year or more. +- **First-person editorial choices the writer can defend.** If the writer can explain _why_ they made a particular cut or used a particular word, that's a strong human signal. +- **Variety in sentence length.** Real writing alternates short and long. AI writing tends toward an even, mid-length cadence. +- **Genuine asides, parentheticals, or self-corrections.** "(I keep wanting to say 'almost' here, but it really was certain.)" Models rarely interrupt themselves like this. +- **Edits made before November 30, 2022.** ChatGPT's public launch. Anything older than that is, with very rare exceptions, not AI-written. + +--- + +## Process and Output + +1. Read the input carefully and identify every instance of the patterns above. +2. Write a **draft rewrite**. Check that it reads naturally aloud, varies sentence length, prefers specific details and simple constructions (is/are/has), and keeps the appropriate register. +3. Ask: **"What makes the below so obviously AI generated?"** Answer briefly with any remaining tells. +4. Revise into a **final rewrite** that addresses them and contains no em or en dashes (see §14). + +Deliver the draft, the brief "still-AI" bullets, the final rewrite, and (optionally) a short summary of changes. + +## Full Example + +**Before (AI-sounding):** + +> I recently spent five unforgettable days in Lisbon, and let me tell you — this city completely stole my heart. From the moment I arrived, I knew I was somewhere truly special. +> +> Nestled along the banks of the Tagus River, Lisbon stands as a vibrant testament to Portugal's enduring spirit, where rich history and modern energy intertwine at every turn. Yes, the famous hills are challenging — my legs certainly felt it! — but every climb rewards you with breathtaking, panoramic views that make it all worthwhile. +> +> No trip would be complete without riding the iconic Tram 28, winding through the city's most historic neighborhoods. And the food? Simply divine. The original pastéis de nata at Pastéis de Belém are a beloved national treasure, and savoring one still warm was a moment I will never forget. +> +> But what truly makes Lisbon special isn't just the sights — it's the feeling. Wander a few steps off the main squares and you'll discover a quieter, more authentic side: sun-drenched alleys, charming tiled facades, and friendly locals going about their daily lives, inviting you to slow down and savor every moment. +> +> Of course, the must-see São Jorge Castle offers stunning views over the rooftops below. The queues can be long, but the experience is absolutely worth it for any history buff or curious traveler. +> +> Would I go back? Absolutely. Lisbon isn't just a place to visit — it's a place to fall in love with, again and again. If you're dreaming of your next getaway, this is one destination that promises memories to last a lifetime. ✨ + +**Draft rewrite:** + +> I spent five days in Lisbon last October and I am still a little split on it. The city is beautiful, but it is also more tiring than I expected. +> +> The hills are not a detail. My hotel was in Alfama, which looked great in photos and meant I had to climb a long staircase every time I wanted coffee. By the second day I was planning walks around which direction went downhill. +> +> I rode Tram 28 because everyone says to ride Tram 28. It was crowded, slow, and mostly full of people filming the same corners. The pastries were better. I had a warm custard tart in Graça that was worth the hype. +> +> The part I liked most was away from the main squares: laundry, tile, open windows, and people watching football in small cafes. The castle view was good, but the line was too long for what it was. +> +> I would go back, but in spring and with better shoes. + +**What makes the below so obviously AI generated?** + +- The draft is too compressed compared with the original; it drops texture instead of rewriting it. +- Some sentences still explain the point too cleanly ("The city is beautiful, but..."). +- The closer is flat and generic rather than a real ending in the same voice. + +**Now make it not obviously AI generated.** + +> I spent five days in Lisbon last October and still have mixed feelings about it. Beautiful, yes. Also harder on the knees than anyone warned me. +> +> The hills are the whole story and somehow never make the brochures. My hotel was up in Alfama, which photographs beautifully and translates, in practice, to climbing what felt like a six-story staircase every time I wanted coffee. By the second day my calves had opinions. I started planning each walk around which way was downhill, which is a strange way to see a city but probably an honest one. +> +> Everyone says to ride Tram 28, so I did, wedged against a stranger's backpack for forty minutes while three tour groups filmed the same corner. I would walk the route next time, or go before breakfast. The custard tarts, though, earn the fuss. I had one at a plain little place in Graça, still warm, and for about thirty seconds I understood why people build trips around pastry. +> +> What I did not expect was how quiet the city gets away from the main squares. Two blocks uphill from any plaza it turns into laundry lines, chipped tile, open windows, and old men watching football with the sound turned up. That is the Lisbon I keep thinking about, not the castle. +> +> The castle is fine. The view is great, the queue is long, and I spent more time shuffling toward the entrance than looking at anything once I got inside. If I had only two days, I would trade it for an afternoon of getting lost. +> +> I would go back, but in spring and with better shoes. Lisbon does not bend over backward to make things easy for you. I think I liked that, even when my legs disagreed. + +**Changes made:** Kept the first-person travel recap and roughly the same level of detail, but removed the chatbot framing, significance inflation, promotional language, forced enthusiasm, em dashes, rule-of-three cadence, generic upbeat conclusion, and emoji. Rebuilt the piece around concrete friction, mixed feelings, uneven rhythm, and specific scenes. + +## Reference + +This skill is based on [Wikipedia:Signs of AI writing](https://en.wikipedia.org/wiki/Wikipedia:Signs_of_AI_writing), maintained by WikiProject AI Cleanup. The patterns documented there come from observations of thousands of instances of AI-generated text on Wikipedia. + +Key insight from Wikipedia: "LLMs use statistical algorithms to guess what should come next. The result tends toward the most statistically likely result that applies to the widest variety of cases." diff --git a/dot_agents/skills/pdf/LICENSE.txt b/dot_agents/skills/pdf/LICENSE.txt new file mode 100644 index 00000000..c55ab422 --- /dev/null +++ b/dot_agents/skills/pdf/LICENSE.txt @@ -0,0 +1,30 @@ +© 2025 Anthropic, PBC. All rights reserved. + +LICENSE: Use of these materials (including all code, prompts, assets, files, +and other components of this Skill) is governed by your agreement with +Anthropic regarding use of Anthropic's services. If no separate agreement +exists, use is governed by Anthropic's Consumer Terms of Service or +Commercial Terms of Service, as applicable: +https://www.anthropic.com/legal/consumer-terms +https://www.anthropic.com/legal/commercial-terms +Your applicable agreement is referred to as the "Agreement." "Services" are +as defined in the Agreement. + +ADDITIONAL RESTRICTIONS: Notwithstanding anything in the Agreement to the +contrary, users may not: + +- Extract these materials from the Services or retain copies of these + materials outside the Services +- Reproduce or copy these materials, except for temporary copies created + automatically during authorized use of the Services +- Create derivative works based on these materials +- Distribute, sublicense, or transfer these materials to any third party +- Make, offer to sell, sell, or import any inventions embodied in these + materials +- Reverse engineer, decompile, or disassemble these materials + +The receipt, viewing, or possession of these materials does not convey or +imply any license or right beyond those expressly granted above. + +Anthropic retains all right, title, and interest in these materials, +including all copyrights, patents, and other intellectual property rights. diff --git a/dot_agents/skills/pdf/SKILL.md b/dot_agents/skills/pdf/SKILL.md new file mode 100644 index 00000000..d3e046a5 --- /dev/null +++ b/dot_agents/skills/pdf/SKILL.md @@ -0,0 +1,314 @@ +--- +name: pdf +description: Use this skill whenever the user wants to do anything with PDF files. This includes reading or extracting text/tables from PDFs, combining or merging multiple PDFs into one, splitting PDFs apart, rotating pages, adding watermarks, creating new PDFs, filling PDF forms, encrypting/decrypting PDFs, extracting images, and OCR on scanned PDFs to make them searchable. If the user mentions a .pdf file or asks to produce one, use this skill. +license: Proprietary. LICENSE.txt has complete terms +--- + +# PDF Processing Guide + +## Overview + +This guide covers essential PDF processing operations using Python libraries and command-line tools. For advanced features, JavaScript libraries, and detailed examples, see REFERENCE.md. If you need to fill out a PDF form, read FORMS.md and follow its instructions. + +## Quick Start + +```python +from pypdf import PdfReader, PdfWriter + +# Read a PDF +reader = PdfReader("document.pdf") +print(f"Pages: {len(reader.pages)}") + +# Extract text +text = "" +for page in reader.pages: + text += page.extract_text() +``` + +## Python Libraries + +### pypdf - Basic Operations + +#### Merge PDFs +```python +from pypdf import PdfWriter, PdfReader + +writer = PdfWriter() +for pdf_file in ["doc1.pdf", "doc2.pdf", "doc3.pdf"]: + reader = PdfReader(pdf_file) + for page in reader.pages: + writer.add_page(page) + +with open("merged.pdf", "wb") as output: + writer.write(output) +``` + +#### Split PDF +```python +reader = PdfReader("input.pdf") +for i, page in enumerate(reader.pages): + writer = PdfWriter() + writer.add_page(page) + with open(f"page_{i+1}.pdf", "wb") as output: + writer.write(output) +``` + +#### Extract Metadata +```python +reader = PdfReader("document.pdf") +meta = reader.metadata +print(f"Title: {meta.title}") +print(f"Author: {meta.author}") +print(f"Subject: {meta.subject}") +print(f"Creator: {meta.creator}") +``` + +#### Rotate Pages +```python +reader = PdfReader("input.pdf") +writer = PdfWriter() + +page = reader.pages[0] +page.rotate(90) # Rotate 90 degrees clockwise +writer.add_page(page) + +with open("rotated.pdf", "wb") as output: + writer.write(output) +``` + +### pdfplumber - Text and Table Extraction + +#### Extract Text with Layout +```python +import pdfplumber + +with pdfplumber.open("document.pdf") as pdf: + for page in pdf.pages: + text = page.extract_text() + print(text) +``` + +#### Extract Tables +```python +with pdfplumber.open("document.pdf") as pdf: + for i, page in enumerate(pdf.pages): + tables = page.extract_tables() + for j, table in enumerate(tables): + print(f"Table {j+1} on page {i+1}:") + for row in table: + print(row) +``` + +#### Advanced Table Extraction +```python +import pandas as pd + +with pdfplumber.open("document.pdf") as pdf: + all_tables = [] + for page in pdf.pages: + tables = page.extract_tables() + for table in tables: + if table: # Check if table is not empty + df = pd.DataFrame(table[1:], columns=table[0]) + all_tables.append(df) + +# Combine all tables +if all_tables: + combined_df = pd.concat(all_tables, ignore_index=True) + combined_df.to_excel("extracted_tables.xlsx", index=False) +``` + +### reportlab - Create PDFs + +#### Basic PDF Creation +```python +from reportlab.lib.pagesizes import letter +from reportlab.pdfgen import canvas + +c = canvas.Canvas("hello.pdf", pagesize=letter) +width, height = letter + +# Add text +c.drawString(100, height - 100, "Hello World!") +c.drawString(100, height - 120, "This is a PDF created with reportlab") + +# Add a line +c.line(100, height - 140, 400, height - 140) + +# Save +c.save() +``` + +#### Create PDF with Multiple Pages +```python +from reportlab.lib.pagesizes import letter +from reportlab.platypus import SimpleDocTemplate, Paragraph, Spacer, PageBreak +from reportlab.lib.styles import getSampleStyleSheet + +doc = SimpleDocTemplate("report.pdf", pagesize=letter) +styles = getSampleStyleSheet() +story = [] + +# Add content +title = Paragraph("Report Title", styles['Title']) +story.append(title) +story.append(Spacer(1, 12)) + +body = Paragraph("This is the body of the report. " * 20, styles['Normal']) +story.append(body) +story.append(PageBreak()) + +# Page 2 +story.append(Paragraph("Page 2", styles['Heading1'])) +story.append(Paragraph("Content for page 2", styles['Normal'])) + +# Build PDF +doc.build(story) +``` + +#### Subscripts and Superscripts + +**IMPORTANT**: Never use Unicode subscript/superscript characters (₀₁₂₃₄₅₆₇₈₉, ⁰¹²³⁴⁵⁶⁷⁸⁹) in ReportLab PDFs. The built-in fonts do not include these glyphs, causing them to render as solid black boxes. + +Instead, use ReportLab's XML markup tags in Paragraph objects: +```python +from reportlab.platypus import Paragraph +from reportlab.lib.styles import getSampleStyleSheet + +styles = getSampleStyleSheet() + +# Subscripts: use tag +chemical = Paragraph("H2O", styles['Normal']) + +# Superscripts: use tag +squared = Paragraph("x2 + y2", styles['Normal']) +``` + +For canvas-drawn text (not Paragraph objects), manually adjust font the size and position rather than using Unicode subscripts/superscripts. + +## Command-Line Tools + +### pdftotext (poppler-utils) +```bash +# Extract text +pdftotext input.pdf output.txt + +# Extract text preserving layout +pdftotext -layout input.pdf output.txt + +# Extract specific pages +pdftotext -f 1 -l 5 input.pdf output.txt # Pages 1-5 +``` + +### qpdf +```bash +# Merge PDFs +qpdf --empty --pages file1.pdf file2.pdf -- merged.pdf + +# Split pages +qpdf input.pdf --pages . 1-5 -- pages1-5.pdf +qpdf input.pdf --pages . 6-10 -- pages6-10.pdf + +# Rotate pages +qpdf input.pdf output.pdf --rotate=+90:1 # Rotate page 1 by 90 degrees + +# Remove password +qpdf --password=mypassword --decrypt encrypted.pdf decrypted.pdf +``` + +### pdftk (if available) +```bash +# Merge +pdftk file1.pdf file2.pdf cat output merged.pdf + +# Split +pdftk input.pdf burst + +# Rotate +pdftk input.pdf rotate 1east output rotated.pdf +``` + +## Common Tasks + +### Extract Text from Scanned PDFs +```python +# Requires: pip install pytesseract pdf2image +import pytesseract +from pdf2image import convert_from_path + +# Convert PDF to images +images = convert_from_path('scanned.pdf') + +# OCR each page +text = "" +for i, image in enumerate(images): + text += f"Page {i+1}:\n" + text += pytesseract.image_to_string(image) + text += "\n\n" + +print(text) +``` + +### Add Watermark +```python +from pypdf import PdfReader, PdfWriter + +# Create watermark (or load existing) +watermark = PdfReader("watermark.pdf").pages[0] + +# Apply to all pages +reader = PdfReader("document.pdf") +writer = PdfWriter() + +for page in reader.pages: + page.merge_page(watermark) + writer.add_page(page) + +with open("watermarked.pdf", "wb") as output: + writer.write(output) +``` + +### Extract Images +```bash +# Using pdfimages (poppler-utils) +pdfimages -j input.pdf output_prefix + +# This extracts all images as output_prefix-000.jpg, output_prefix-001.jpg, etc. +``` + +### Password Protection +```python +from pypdf import PdfReader, PdfWriter + +reader = PdfReader("input.pdf") +writer = PdfWriter() + +for page in reader.pages: + writer.add_page(page) + +# Add password +writer.encrypt("userpassword", "ownerpassword") + +with open("encrypted.pdf", "wb") as output: + writer.write(output) +``` + +## Quick Reference + +| Task | Best Tool | Command/Code | +|------|-----------|--------------| +| Merge PDFs | pypdf | `writer.add_page(page)` | +| Split PDFs | pypdf | One page per file | +| Extract text | pdfplumber | `page.extract_text()` | +| Extract tables | pdfplumber | `page.extract_tables()` | +| Create PDFs | reportlab | Canvas or Platypus | +| Command line merge | qpdf | `qpdf --empty --pages ...` | +| OCR scanned PDFs | pytesseract | Convert to image first | +| Fill PDF forms | pdf-lib or pypdf (see FORMS.md) | See FORMS.md | + +## Next Steps + +- For advanced pypdfium2 usage, see REFERENCE.md +- For JavaScript libraries (pdf-lib), see REFERENCE.md +- If you need to fill out a PDF form, follow the instructions in FORMS.md +- For troubleshooting guides, see REFERENCE.md diff --git a/dot_agents/skills/pdf/forms.md b/dot_agents/skills/pdf/forms.md new file mode 100644 index 00000000..6e7e1e0d --- /dev/null +++ b/dot_agents/skills/pdf/forms.md @@ -0,0 +1,294 @@ +**CRITICAL: You MUST complete these steps in order. Do not skip ahead to writing code.** + +If you need to fill out a PDF form, first check to see if the PDF has fillable form fields. Run this script from this file's directory: + `python scripts/check_fillable_fields `, and depending on the result go to either the "Fillable fields" or "Non-fillable fields" and follow those instructions. + +# Fillable fields +If the PDF has fillable form fields: +- Run this script from this file's directory: `python scripts/extract_form_field_info.py `. It will create a JSON file with a list of fields in this format: +``` +[ + { + "field_id": (unique ID for the field), + "page": (page number, 1-based), + "rect": ([left, bottom, right, top] bounding box in PDF coordinates, y=0 is the bottom of the page), + "type": ("text", "checkbox", "radio_group", or "choice"), + }, + // Checkboxes have "checked_value" and "unchecked_value" properties: + { + "field_id": (unique ID for the field), + "page": (page number, 1-based), + "type": "checkbox", + "checked_value": (Set the field to this value to check the checkbox), + "unchecked_value": (Set the field to this value to uncheck the checkbox), + }, + // Radio groups have a "radio_options" list with the possible choices. + { + "field_id": (unique ID for the field), + "page": (page number, 1-based), + "type": "radio_group", + "radio_options": [ + { + "value": (set the field to this value to select this radio option), + "rect": (bounding box for the radio button for this option) + }, + // Other radio options + ] + }, + // Multiple choice fields have a "choice_options" list with the possible choices: + { + "field_id": (unique ID for the field), + "page": (page number, 1-based), + "type": "choice", + "choice_options": [ + { + "value": (set the field to this value to select this option), + "text": (display text of the option) + }, + // Other choice options + ], + } +] +``` +- Convert the PDF to PNGs (one image for each page) with this script (run from this file's directory): +`python scripts/convert_pdf_to_images.py ` +Then analyze the images to determine the purpose of each form field (make sure to convert the bounding box PDF coordinates to image coordinates). +- Create a `field_values.json` file in this format with the values to be entered for each field: +``` +[ + { + "field_id": "last_name", // Must match the field_id from `extract_form_field_info.py` + "description": "The user's last name", + "page": 1, // Must match the "page" value in field_info.json + "value": "Simpson" + }, + { + "field_id": "Checkbox12", + "description": "Checkbox to be checked if the user is 18 or over", + "page": 1, + "value": "/On" // If this is a checkbox, use its "checked_value" value to check it. If it's a radio button group, use one of the "value" values in "radio_options". + }, + // more fields +] +``` +- Run the `fill_fillable_fields.py` script from this file's directory to create a filled-in PDF: +`python scripts/fill_fillable_fields.py ` +This script will verify that the field IDs and values you provide are valid; if it prints error messages, correct the appropriate fields and try again. + +# Non-fillable fields +If the PDF doesn't have fillable form fields, you'll add text annotations. First try to extract coordinates from the PDF structure (more accurate), then fall back to visual estimation if needed. + +## Step 1: Try Structure Extraction First + +Run this script to extract text labels, lines, and checkboxes with their exact PDF coordinates: +`python scripts/extract_form_structure.py form_structure.json` + +This creates a JSON file containing: +- **labels**: Every text element with exact coordinates (x0, top, x1, bottom in PDF points) +- **lines**: Horizontal lines that define row boundaries +- **checkboxes**: Small square rectangles that are checkboxes (with center coordinates) +- **row_boundaries**: Row top/bottom positions calculated from horizontal lines + +**Check the results**: If `form_structure.json` has meaningful labels (text elements that correspond to form fields), use **Approach A: Structure-Based Coordinates**. If the PDF is scanned/image-based and has few or no labels, use **Approach B: Visual Estimation**. + +--- + +## Approach A: Structure-Based Coordinates (Preferred) + +Use this when `extract_form_structure.py` found text labels in the PDF. + +### A.1: Analyze the Structure + +Read form_structure.json and identify: + +1. **Label groups**: Adjacent text elements that form a single label (e.g., "Last" + "Name") +2. **Row structure**: Labels with similar `top` values are in the same row +3. **Field columns**: Entry areas start after label ends (x0 = label.x1 + gap) +4. **Checkboxes**: Use the checkbox coordinates directly from the structure + +**Coordinate system**: PDF coordinates where y=0 is at TOP of page, y increases downward. + +### A.2: Check for Missing Elements + +The structure extraction may not detect all form elements. Common cases: +- **Circular checkboxes**: Only square rectangles are detected as checkboxes +- **Complex graphics**: Decorative elements or non-standard form controls +- **Faded or light-colored elements**: May not be extracted + +If you see form fields in the PDF images that aren't in form_structure.json, you'll need to use **visual analysis** for those specific fields (see "Hybrid Approach" below). + +### A.3: Create fields.json with PDF Coordinates + +For each field, calculate entry coordinates from the extracted structure: + +**Text fields:** +- entry x0 = label x1 + 5 (small gap after label) +- entry x1 = next label's x0, or row boundary +- entry top = same as label top +- entry bottom = row boundary line below, or label bottom + row_height + +**Checkboxes:** +- Use the checkbox rectangle coordinates directly from form_structure.json +- entry_bounding_box = [checkbox.x0, checkbox.top, checkbox.x1, checkbox.bottom] + +Create fields.json using `pdf_width` and `pdf_height` (signals PDF coordinates): +```json +{ + "pages": [ + {"page_number": 1, "pdf_width": 612, "pdf_height": 792} + ], + "form_fields": [ + { + "page_number": 1, + "description": "Last name entry field", + "field_label": "Last Name", + "label_bounding_box": [43, 63, 87, 73], + "entry_bounding_box": [92, 63, 260, 79], + "entry_text": {"text": "Smith", "font_size": 10} + }, + { + "page_number": 1, + "description": "US Citizen Yes checkbox", + "field_label": "Yes", + "label_bounding_box": [260, 200, 280, 210], + "entry_bounding_box": [285, 197, 292, 205], + "entry_text": {"text": "X"} + } + ] +} +``` + +**Important**: Use `pdf_width`/`pdf_height` and coordinates directly from form_structure.json. + +### A.4: Validate Bounding Boxes + +Before filling, check your bounding boxes for errors: +`python scripts/check_bounding_boxes.py fields.json` + +This checks for intersecting bounding boxes and entry boxes that are too small for the font size. Fix any reported errors before filling. + +--- + +## Approach B: Visual Estimation (Fallback) + +Use this when the PDF is scanned/image-based and structure extraction found no usable text labels (e.g., all text shows as "(cid:X)" patterns). + +### B.1: Convert PDF to Images + +`python scripts/convert_pdf_to_images.py ` + +### B.2: Initial Field Identification + +Examine each page image to identify form sections and get **rough estimates** of field locations: +- Form field labels and their approximate positions +- Entry areas (lines, boxes, or blank spaces for text input) +- Checkboxes and their approximate locations + +For each field, note approximate pixel coordinates (they don't need to be precise yet). + +### B.3: Zoom Refinement (CRITICAL for accuracy) + +For each field, crop a region around the estimated position to refine coordinates precisely. + +**Create a zoomed crop using ImageMagick:** +```bash +magick -crop x++ +repage +``` + +Where: +- `, ` = top-left corner of crop region (use your rough estimate minus padding) +- `, ` = size of crop region (field area plus ~50px padding on each side) + +**Example:** To refine a "Name" field estimated around (100, 150): +```bash +magick images_dir/page_1.png -crop 300x80+50+120 +repage crops/name_field.png +``` + +(Note: if the `magick` command isn't available, try `convert` with the same arguments). + +**Examine the cropped image** to determine precise coordinates: +1. Identify the exact pixel where the entry area begins (after the label) +2. Identify where the entry area ends (before next field or edge) +3. Identify the top and bottom of the entry line/box + +**Convert crop coordinates back to full image coordinates:** +- full_x = crop_x + crop_offset_x +- full_y = crop_y + crop_offset_y + +Example: If the crop started at (50, 120) and the entry box starts at (52, 18) within the crop: +- entry_x0 = 52 + 50 = 102 +- entry_top = 18 + 120 = 138 + +**Repeat for each field**, grouping nearby fields into single crops when possible. + +### B.4: Create fields.json with Refined Coordinates + +Create fields.json using `image_width` and `image_height` (signals image coordinates): +```json +{ + "pages": [ + {"page_number": 1, "image_width": 1700, "image_height": 2200} + ], + "form_fields": [ + { + "page_number": 1, + "description": "Last name entry field", + "field_label": "Last Name", + "label_bounding_box": [120, 175, 242, 198], + "entry_bounding_box": [255, 175, 720, 218], + "entry_text": {"text": "Smith", "font_size": 10} + } + ] +} +``` + +**Important**: Use `image_width`/`image_height` and the refined pixel coordinates from the zoom analysis. + +### B.5: Validate Bounding Boxes + +Before filling, check your bounding boxes for errors: +`python scripts/check_bounding_boxes.py fields.json` + +This checks for intersecting bounding boxes and entry boxes that are too small for the font size. Fix any reported errors before filling. + +--- + +## Hybrid Approach: Structure + Visual + +Use this when structure extraction works for most fields but misses some elements (e.g., circular checkboxes, unusual form controls). + +1. **Use Approach A** for fields that were detected in form_structure.json +2. **Convert PDF to images** for visual analysis of missing fields +3. **Use zoom refinement** (from Approach B) for the missing fields +4. **Combine coordinates**: For fields from structure extraction, use `pdf_width`/`pdf_height`. For visually-estimated fields, you must convert image coordinates to PDF coordinates: + - pdf_x = image_x * (pdf_width / image_width) + - pdf_y = image_y * (pdf_height / image_height) +5. **Use a single coordinate system** in fields.json - convert all to PDF coordinates with `pdf_width`/`pdf_height` + +--- + +## Step 2: Validate Before Filling + +**Always validate bounding boxes before filling:** +`python scripts/check_bounding_boxes.py fields.json` + +This checks for: +- Intersecting bounding boxes (which would cause overlapping text) +- Entry boxes that are too small for the specified font size + +Fix any reported errors in fields.json before proceeding. + +## Step 3: Fill the Form + +The fill script auto-detects the coordinate system and handles conversion: +`python scripts/fill_pdf_form_with_annotations.py fields.json ` + +## Step 4: Verify Output + +Convert the filled PDF to images and verify text placement: +`python scripts/convert_pdf_to_images.py ` + +If text is mispositioned: +- **Approach A**: Check that you're using PDF coordinates from form_structure.json with `pdf_width`/`pdf_height` +- **Approach B**: Check that image dimensions match and coordinates are accurate pixels +- **Hybrid**: Ensure coordinate conversions are correct for visually-estimated fields diff --git a/dot_agents/skills/pdf/reference.md b/dot_agents/skills/pdf/reference.md new file mode 100644 index 00000000..41400bf4 --- /dev/null +++ b/dot_agents/skills/pdf/reference.md @@ -0,0 +1,612 @@ +# PDF Processing Advanced Reference + +This document contains advanced PDF processing features, detailed examples, and additional libraries not covered in the main skill instructions. + +## pypdfium2 Library (Apache/BSD License) + +### Overview +pypdfium2 is a Python binding for PDFium (Chromium's PDF library). It's excellent for fast PDF rendering, image generation, and serves as a PyMuPDF replacement. + +### Render PDF to Images +```python +import pypdfium2 as pdfium +from PIL import Image + +# Load PDF +pdf = pdfium.PdfDocument("document.pdf") + +# Render page to image +page = pdf[0] # First page +bitmap = page.render( + scale=2.0, # Higher resolution + rotation=0 # No rotation +) + +# Convert to PIL Image +img = bitmap.to_pil() +img.save("page_1.png", "PNG") + +# Process multiple pages +for i, page in enumerate(pdf): + bitmap = page.render(scale=1.5) + img = bitmap.to_pil() + img.save(f"page_{i+1}.jpg", "JPEG", quality=90) +``` + +### Extract Text with pypdfium2 +```python +import pypdfium2 as pdfium + +pdf = pdfium.PdfDocument("document.pdf") +for i, page in enumerate(pdf): + text = page.get_text() + print(f"Page {i+1} text length: {len(text)} chars") +``` + +## JavaScript Libraries + +### pdf-lib (MIT License) + +pdf-lib is a powerful JavaScript library for creating and modifying PDF documents in any JavaScript environment. + +#### Load and Manipulate Existing PDF +```javascript +import { PDFDocument } from 'pdf-lib'; +import fs from 'fs'; + +async function manipulatePDF() { + // Load existing PDF + const existingPdfBytes = fs.readFileSync('input.pdf'); + const pdfDoc = await PDFDocument.load(existingPdfBytes); + + // Get page count + const pageCount = pdfDoc.getPageCount(); + console.log(`Document has ${pageCount} pages`); + + // Add new page + const newPage = pdfDoc.addPage([600, 400]); + newPage.drawText('Added by pdf-lib', { + x: 100, + y: 300, + size: 16 + }); + + // Save modified PDF + const pdfBytes = await pdfDoc.save(); + fs.writeFileSync('modified.pdf', pdfBytes); +} +``` + +#### Create Complex PDFs from Scratch +```javascript +import { PDFDocument, rgb, StandardFonts } from 'pdf-lib'; +import fs from 'fs'; + +async function createPDF() { + const pdfDoc = await PDFDocument.create(); + + // Add fonts + const helveticaFont = await pdfDoc.embedFont(StandardFonts.Helvetica); + const helveticaBold = await pdfDoc.embedFont(StandardFonts.HelveticaBold); + + // Add page + const page = pdfDoc.addPage([595, 842]); // A4 size + const { width, height } = page.getSize(); + + // Add text with styling + page.drawText('Invoice #12345', { + x: 50, + y: height - 50, + size: 18, + font: helveticaBold, + color: rgb(0.2, 0.2, 0.8) + }); + + // Add rectangle (header background) + page.drawRectangle({ + x: 40, + y: height - 100, + width: width - 80, + height: 30, + color: rgb(0.9, 0.9, 0.9) + }); + + // Add table-like content + const items = [ + ['Item', 'Qty', 'Price', 'Total'], + ['Widget', '2', '$50', '$100'], + ['Gadget', '1', '$75', '$75'] + ]; + + let yPos = height - 150; + items.forEach(row => { + let xPos = 50; + row.forEach(cell => { + page.drawText(cell, { + x: xPos, + y: yPos, + size: 12, + font: helveticaFont + }); + xPos += 120; + }); + yPos -= 25; + }); + + const pdfBytes = await pdfDoc.save(); + fs.writeFileSync('created.pdf', pdfBytes); +} +``` + +#### Advanced Merge and Split Operations +```javascript +import { PDFDocument } from 'pdf-lib'; +import fs from 'fs'; + +async function mergePDFs() { + // Create new document + const mergedPdf = await PDFDocument.create(); + + // Load source PDFs + const pdf1Bytes = fs.readFileSync('doc1.pdf'); + const pdf2Bytes = fs.readFileSync('doc2.pdf'); + + const pdf1 = await PDFDocument.load(pdf1Bytes); + const pdf2 = await PDFDocument.load(pdf2Bytes); + + // Copy pages from first PDF + const pdf1Pages = await mergedPdf.copyPages(pdf1, pdf1.getPageIndices()); + pdf1Pages.forEach(page => mergedPdf.addPage(page)); + + // Copy specific pages from second PDF (pages 0, 2, 4) + const pdf2Pages = await mergedPdf.copyPages(pdf2, [0, 2, 4]); + pdf2Pages.forEach(page => mergedPdf.addPage(page)); + + const mergedPdfBytes = await mergedPdf.save(); + fs.writeFileSync('merged.pdf', mergedPdfBytes); +} +``` + +### pdfjs-dist (Apache License) + +PDF.js is Mozilla's JavaScript library for rendering PDFs in the browser. + +#### Basic PDF Loading and Rendering +```javascript +import * as pdfjsLib from 'pdfjs-dist'; + +// Configure worker (important for performance) +pdfjsLib.GlobalWorkerOptions.workerSrc = './pdf.worker.js'; + +async function renderPDF() { + // Load PDF + const loadingTask = pdfjsLib.getDocument('document.pdf'); + const pdf = await loadingTask.promise; + + console.log(`Loaded PDF with ${pdf.numPages} pages`); + + // Get first page + const page = await pdf.getPage(1); + const viewport = page.getViewport({ scale: 1.5 }); + + // Render to canvas + const canvas = document.createElement('canvas'); + const context = canvas.getContext('2d'); + canvas.height = viewport.height; + canvas.width = viewport.width; + + const renderContext = { + canvasContext: context, + viewport: viewport + }; + + await page.render(renderContext).promise; + document.body.appendChild(canvas); +} +``` + +#### Extract Text with Coordinates +```javascript +import * as pdfjsLib from 'pdfjs-dist'; + +async function extractText() { + const loadingTask = pdfjsLib.getDocument('document.pdf'); + const pdf = await loadingTask.promise; + + let fullText = ''; + + // Extract text from all pages + for (let i = 1; i <= pdf.numPages; i++) { + const page = await pdf.getPage(i); + const textContent = await page.getTextContent(); + + const pageText = textContent.items + .map(item => item.str) + .join(' '); + + fullText += `\n--- Page ${i} ---\n${pageText}`; + + // Get text with coordinates for advanced processing + const textWithCoords = textContent.items.map(item => ({ + text: item.str, + x: item.transform[4], + y: item.transform[5], + width: item.width, + height: item.height + })); + } + + console.log(fullText); + return fullText; +} +``` + +#### Extract Annotations and Forms +```javascript +import * as pdfjsLib from 'pdfjs-dist'; + +async function extractAnnotations() { + const loadingTask = pdfjsLib.getDocument('annotated.pdf'); + const pdf = await loadingTask.promise; + + for (let i = 1; i <= pdf.numPages; i++) { + const page = await pdf.getPage(i); + const annotations = await page.getAnnotations(); + + annotations.forEach(annotation => { + console.log(`Annotation type: ${annotation.subtype}`); + console.log(`Content: ${annotation.contents}`); + console.log(`Coordinates: ${JSON.stringify(annotation.rect)}`); + }); + } +} +``` + +## Advanced Command-Line Operations + +### poppler-utils Advanced Features + +#### Extract Text with Bounding Box Coordinates +```bash +# Extract text with bounding box coordinates (essential for structured data) +pdftotext -bbox-layout document.pdf output.xml + +# The XML output contains precise coordinates for each text element +``` + +#### Advanced Image Conversion +```bash +# Convert to PNG images with specific resolution +pdftoppm -png -r 300 document.pdf output_prefix + +# Convert specific page range with high resolution +pdftoppm -png -r 600 -f 1 -l 3 document.pdf high_res_pages + +# Convert to JPEG with quality setting +pdftoppm -jpeg -jpegopt quality=85 -r 200 document.pdf jpeg_output +``` + +#### Extract Embedded Images +```bash +# Extract all embedded images with metadata +pdfimages -j -p document.pdf page_images + +# List image info without extracting +pdfimages -list document.pdf + +# Extract images in their original format +pdfimages -all document.pdf images/img +``` + +### qpdf Advanced Features + +#### Complex Page Manipulation +```bash +# Split PDF into groups of pages +qpdf --split-pages=3 input.pdf output_group_%02d.pdf + +# Extract specific pages with complex ranges +qpdf input.pdf --pages input.pdf 1,3-5,8,10-end -- extracted.pdf + +# Merge specific pages from multiple PDFs +qpdf --empty --pages doc1.pdf 1-3 doc2.pdf 5-7 doc3.pdf 2,4 -- combined.pdf +``` + +#### PDF Optimization and Repair +```bash +# Optimize PDF for web (linearize for streaming) +qpdf --linearize input.pdf optimized.pdf + +# Remove unused objects and compress +qpdf --optimize-level=all input.pdf compressed.pdf + +# Attempt to repair corrupted PDF structure +qpdf --check input.pdf +qpdf --fix-qdf damaged.pdf repaired.pdf + +# Show detailed PDF structure for debugging +qpdf --show-all-pages input.pdf > structure.txt +``` + +#### Advanced Encryption +```bash +# Add password protection with specific permissions +qpdf --encrypt user_pass owner_pass 256 --print=none --modify=none -- input.pdf encrypted.pdf + +# Check encryption status +qpdf --show-encryption encrypted.pdf + +# Remove password protection (requires password) +qpdf --password=secret123 --decrypt encrypted.pdf decrypted.pdf +``` + +## Advanced Python Techniques + +### pdfplumber Advanced Features + +#### Extract Text with Precise Coordinates +```python +import pdfplumber + +with pdfplumber.open("document.pdf") as pdf: + page = pdf.pages[0] + + # Extract all text with coordinates + chars = page.chars + for char in chars[:10]: # First 10 characters + print(f"Char: '{char['text']}' at x:{char['x0']:.1f} y:{char['y0']:.1f}") + + # Extract text by bounding box (left, top, right, bottom) + bbox_text = page.within_bbox((100, 100, 400, 200)).extract_text() +``` + +#### Advanced Table Extraction with Custom Settings +```python +import pdfplumber +import pandas as pd + +with pdfplumber.open("complex_table.pdf") as pdf: + page = pdf.pages[0] + + # Extract tables with custom settings for complex layouts + table_settings = { + "vertical_strategy": "lines", + "horizontal_strategy": "lines", + "snap_tolerance": 3, + "intersection_tolerance": 15 + } + tables = page.extract_tables(table_settings) + + # Visual debugging for table extraction + img = page.to_image(resolution=150) + img.save("debug_layout.png") +``` + +### reportlab Advanced Features + +#### Create Professional Reports with Tables +```python +from reportlab.platypus import SimpleDocTemplate, Table, TableStyle, Paragraph +from reportlab.lib.styles import getSampleStyleSheet +from reportlab.lib import colors + +# Sample data +data = [ + ['Product', 'Q1', 'Q2', 'Q3', 'Q4'], + ['Widgets', '120', '135', '142', '158'], + ['Gadgets', '85', '92', '98', '105'] +] + +# Create PDF with table +doc = SimpleDocTemplate("report.pdf") +elements = [] + +# Add title +styles = getSampleStyleSheet() +title = Paragraph("Quarterly Sales Report", styles['Title']) +elements.append(title) + +# Add table with advanced styling +table = Table(data) +table.setStyle(TableStyle([ + ('BACKGROUND', (0, 0), (-1, 0), colors.grey), + ('TEXTCOLOR', (0, 0), (-1, 0), colors.whitesmoke), + ('ALIGN', (0, 0), (-1, -1), 'CENTER'), + ('FONTNAME', (0, 0), (-1, 0), 'Helvetica-Bold'), + ('FONTSIZE', (0, 0), (-1, 0), 14), + ('BOTTOMPADDING', (0, 0), (-1, 0), 12), + ('BACKGROUND', (0, 1), (-1, -1), colors.beige), + ('GRID', (0, 0), (-1, -1), 1, colors.black) +])) +elements.append(table) + +doc.build(elements) +``` + +## Complex Workflows + +### Extract Figures/Images from PDF + +#### Method 1: Using pdfimages (fastest) +```bash +# Extract all images with original quality +pdfimages -all document.pdf images/img +``` + +#### Method 2: Using pypdfium2 + Image Processing +```python +import pypdfium2 as pdfium +from PIL import Image +import numpy as np + +def extract_figures(pdf_path, output_dir): + pdf = pdfium.PdfDocument(pdf_path) + + for page_num, page in enumerate(pdf): + # Render high-resolution page + bitmap = page.render(scale=3.0) + img = bitmap.to_pil() + + # Convert to numpy for processing + img_array = np.array(img) + + # Simple figure detection (non-white regions) + mask = np.any(img_array != [255, 255, 255], axis=2) + + # Find contours and extract bounding boxes + # (This is simplified - real implementation would need more sophisticated detection) + + # Save detected figures + # ... implementation depends on specific needs +``` + +### Batch PDF Processing with Error Handling +```python +import os +import glob +from pypdf import PdfReader, PdfWriter +import logging + +logging.basicConfig(level=logging.INFO) +logger = logging.getLogger(__name__) + +def batch_process_pdfs(input_dir, operation='merge'): + pdf_files = glob.glob(os.path.join(input_dir, "*.pdf")) + + if operation == 'merge': + writer = PdfWriter() + for pdf_file in pdf_files: + try: + reader = PdfReader(pdf_file) + for page in reader.pages: + writer.add_page(page) + logger.info(f"Processed: {pdf_file}") + except Exception as e: + logger.error(f"Failed to process {pdf_file}: {e}") + continue + + with open("batch_merged.pdf", "wb") as output: + writer.write(output) + + elif operation == 'extract_text': + for pdf_file in pdf_files: + try: + reader = PdfReader(pdf_file) + text = "" + for page in reader.pages: + text += page.extract_text() + + output_file = pdf_file.replace('.pdf', '.txt') + with open(output_file, 'w', encoding='utf-8') as f: + f.write(text) + logger.info(f"Extracted text from: {pdf_file}") + + except Exception as e: + logger.error(f"Failed to extract text from {pdf_file}: {e}") + continue +``` + +### Advanced PDF Cropping +```python +from pypdf import PdfWriter, PdfReader + +reader = PdfReader("input.pdf") +writer = PdfWriter() + +# Crop page (left, bottom, right, top in points) +page = reader.pages[0] +page.mediabox.left = 50 +page.mediabox.bottom = 50 +page.mediabox.right = 550 +page.mediabox.top = 750 + +writer.add_page(page) +with open("cropped.pdf", "wb") as output: + writer.write(output) +``` + +## Performance Optimization Tips + +### 1. For Large PDFs +- Use streaming approaches instead of loading entire PDF in memory +- Use `qpdf --split-pages` for splitting large files +- Process pages individually with pypdfium2 + +### 2. For Text Extraction +- `pdftotext -bbox-layout` is fastest for plain text extraction +- Use pdfplumber for structured data and tables +- Avoid `pypdf.extract_text()` for very large documents + +### 3. For Image Extraction +- `pdfimages` is much faster than rendering pages +- Use low resolution for previews, high resolution for final output + +### 4. For Form Filling +- pdf-lib maintains form structure better than most alternatives +- Pre-validate form fields before processing + +### 5. Memory Management +```python +# Process PDFs in chunks +def process_large_pdf(pdf_path, chunk_size=10): + reader = PdfReader(pdf_path) + total_pages = len(reader.pages) + + for start_idx in range(0, total_pages, chunk_size): + end_idx = min(start_idx + chunk_size, total_pages) + writer = PdfWriter() + + for i in range(start_idx, end_idx): + writer.add_page(reader.pages[i]) + + # Process chunk + with open(f"chunk_{start_idx//chunk_size}.pdf", "wb") as output: + writer.write(output) +``` + +## Troubleshooting Common Issues + +### Encrypted PDFs +```python +# Handle password-protected PDFs +from pypdf import PdfReader + +try: + reader = PdfReader("encrypted.pdf") + if reader.is_encrypted: + reader.decrypt("password") +except Exception as e: + print(f"Failed to decrypt: {e}") +``` + +### Corrupted PDFs +```bash +# Use qpdf to repair +qpdf --check corrupted.pdf +qpdf --replace-input corrupted.pdf +``` + +### Text Extraction Issues +```python +# Fallback to OCR for scanned PDFs +import pytesseract +from pdf2image import convert_from_path + +def extract_text_with_ocr(pdf_path): + images = convert_from_path(pdf_path) + text = "" + for i, image in enumerate(images): + text += pytesseract.image_to_string(image) + return text +``` + +## License Information + +- **pypdf**: BSD License +- **pdfplumber**: MIT License +- **pypdfium2**: Apache/BSD License +- **reportlab**: BSD License +- **poppler-utils**: GPL-2 License +- **qpdf**: Apache License +- **pdf-lib**: MIT License +- **pdfjs-dist**: Apache License \ No newline at end of file diff --git a/dot_agents/skills/pdf/scripts/check_bounding_boxes.py b/dot_agents/skills/pdf/scripts/check_bounding_boxes.py new file mode 100644 index 00000000..eea39186 --- /dev/null +++ b/dot_agents/skills/pdf/scripts/check_bounding_boxes.py @@ -0,0 +1,76 @@ +from dataclasses import dataclass +import json +import sys + + +@dataclass +class RectAndField: + rect: list[float] + rect_type: str + field: dict + + +def get_bounding_box_messages(fields_json_stream) -> list[str]: + messages = [] + fields = json.load(fields_json_stream) + messages.append(f"Read {len(fields['form_fields'])} fields") + + def rects_intersect(r1, r2): + disjoint_horizontal = r1[0] >= r2[2] or r1[2] <= r2[0] + disjoint_vertical = r1[1] >= r2[3] or r1[3] <= r2[1] + return not (disjoint_horizontal or disjoint_vertical) + + rects_and_fields = [] + for f in fields["form_fields"]: + rects_and_fields.append(RectAndField(f["label_bounding_box"], "label", f)) + rects_and_fields.append(RectAndField(f["entry_bounding_box"], "entry", f)) + + has_error = False + for i, ri in enumerate(rects_and_fields): + for j in range(i + 1, len(rects_and_fields)): + rj = rects_and_fields[j] + if ri.field["page_number"] == rj.field["page_number"] and rects_intersect( + ri.rect, rj.rect + ): + has_error = True + if ri.field is rj.field: + messages.append( + f"FAILURE: intersection between label and entry bounding boxes for `{ri.field['description']}` ({ri.rect}, {rj.rect})" + ) + else: + messages.append( + f"FAILURE: intersection between {ri.rect_type} bounding box for `{ri.field['description']}` ({ri.rect}) and {rj.rect_type} bounding box for `{rj.field['description']}` ({rj.rect})" + ) + if len(messages) >= 20: + messages.append( + "Aborting further checks; fix bounding boxes and try again" + ) + return messages + if ri.rect_type == "entry": + if "entry_text" in ri.field: + font_size = ri.field["entry_text"].get("font_size", 14) + entry_height = ri.rect[3] - ri.rect[1] + if entry_height < font_size: + has_error = True + messages.append( + f"FAILURE: entry bounding box height ({entry_height}) for `{ri.field['description']}` is too short for the text content (font size: {font_size}). Increase the box height or decrease the font size." + ) + if len(messages) >= 20: + messages.append( + "Aborting further checks; fix bounding boxes and try again" + ) + return messages + + if not has_error: + messages.append("SUCCESS: All bounding boxes are valid") + return messages + + +if __name__ == "__main__": + if len(sys.argv) != 2: + print("Usage: check_bounding_boxes.py [fields.json]") + sys.exit(1) + with open(sys.argv[1]) as f: + messages = get_bounding_box_messages(f) + for msg in messages: + print(msg) diff --git a/dot_agents/skills/pdf/scripts/check_fillable_fields.py b/dot_agents/skills/pdf/scripts/check_fillable_fields.py new file mode 100644 index 00000000..c32f9e1b --- /dev/null +++ b/dot_agents/skills/pdf/scripts/check_fillable_fields.py @@ -0,0 +1,10 @@ +import sys +from pypdf import PdfReader + +reader = PdfReader(sys.argv[1]) +if reader.get_fields(): + print("This PDF has fillable form fields") +else: + print( + "This PDF does not have fillable form fields; you will need to visually determine where to enter data" + ) diff --git a/dot_agents/skills/pdf/scripts/convert_pdf_to_images.py b/dot_agents/skills/pdf/scripts/convert_pdf_to_images.py new file mode 100644 index 00000000..c35a7606 --- /dev/null +++ b/dot_agents/skills/pdf/scripts/convert_pdf_to_images.py @@ -0,0 +1,31 @@ +import os +import sys + +from pdf2image import convert_from_path + + +def convert(pdf_path, output_dir, max_dim=1000): + images = convert_from_path(pdf_path, dpi=200) + + for i, image in enumerate(images): + width, height = image.size + if width > max_dim or height > max_dim: + scale_factor = min(max_dim / width, max_dim / height) + new_width = int(width * scale_factor) + new_height = int(height * scale_factor) + image = image.resize((new_width, new_height)) + + image_path = os.path.join(output_dir, f"page_{i+1}.png") + image.save(image_path) + print(f"Saved page {i+1} as {image_path} (size: {image.size})") + + print(f"Converted {len(images)} pages to PNG images") + + +if __name__ == "__main__": + if len(sys.argv) != 3: + print("Usage: convert_pdf_to_images.py [input pdf] [output directory]") + sys.exit(1) + pdf_path = sys.argv[1] + output_directory = sys.argv[2] + convert(pdf_path, output_directory) diff --git a/dot_agents/skills/pdf/scripts/executable_extract_form_structure.py b/dot_agents/skills/pdf/scripts/executable_extract_form_structure.py new file mode 100644 index 00000000..7b2581e2 --- /dev/null +++ b/dot_agents/skills/pdf/scripts/executable_extract_form_structure.py @@ -0,0 +1,129 @@ +""" +Extract form structure from a non-fillable PDF. + +This script analyzes the PDF to find: +- Text labels with their exact coordinates +- Horizontal lines (row boundaries) +- Checkboxes (small rectangles) + +Output: A JSON file with the form structure that can be used to generate +accurate field coordinates for filling. + +Usage: python extract_form_structure.py +""" + +import json +import sys +import pdfplumber + + +def extract_form_structure(pdf_path): + structure = { + "pages": [], + "labels": [], + "lines": [], + "checkboxes": [], + "row_boundaries": [], + } + + with pdfplumber.open(pdf_path) as pdf: + for page_num, page in enumerate(pdf.pages, 1): + structure["pages"].append( + { + "page_number": page_num, + "width": float(page.width), + "height": float(page.height), + } + ) + + words = page.extract_words() + for word in words: + structure["labels"].append( + { + "page": page_num, + "text": word["text"], + "x0": round(float(word["x0"]), 1), + "top": round(float(word["top"]), 1), + "x1": round(float(word["x1"]), 1), + "bottom": round(float(word["bottom"]), 1), + } + ) + + for line in page.lines: + if abs(float(line["x1"]) - float(line["x0"])) > page.width * 0.5: + structure["lines"].append( + { + "page": page_num, + "y": round(float(line["top"]), 1), + "x0": round(float(line["x0"]), 1), + "x1": round(float(line["x1"]), 1), + } + ) + + for rect in page.rects: + width = float(rect["x1"]) - float(rect["x0"]) + height = float(rect["bottom"]) - float(rect["top"]) + if 5 <= width <= 15 and 5 <= height <= 15 and abs(width - height) < 2: + structure["checkboxes"].append( + { + "page": page_num, + "x0": round(float(rect["x0"]), 1), + "top": round(float(rect["top"]), 1), + "x1": round(float(rect["x1"]), 1), + "bottom": round(float(rect["bottom"]), 1), + "center_x": round( + (float(rect["x0"]) + float(rect["x1"])) / 2, 1 + ), + "center_y": round( + (float(rect["top"]) + float(rect["bottom"])) / 2, 1 + ), + } + ) + + lines_by_page = {} + for line in structure["lines"]: + page = line["page"] + if page not in lines_by_page: + lines_by_page[page] = [] + lines_by_page[page].append(line["y"]) + + for page, y_coords in lines_by_page.items(): + y_coords = sorted(set(y_coords)) + for i in range(len(y_coords) - 1): + structure["row_boundaries"].append( + { + "page": page, + "row_top": y_coords[i], + "row_bottom": y_coords[i + 1], + "row_height": round(y_coords[i + 1] - y_coords[i], 1), + } + ) + + return structure + + +def main(): + if len(sys.argv) != 3: + print("Usage: extract_form_structure.py ") + sys.exit(1) + + pdf_path = sys.argv[1] + output_path = sys.argv[2] + + print(f"Extracting structure from {pdf_path}...") + structure = extract_form_structure(pdf_path) + + with open(output_path, "w") as f: + json.dump(structure, f, indent=2) + + print(f"Found:") + print(f" - {len(structure['pages'])} pages") + print(f" - {len(structure['labels'])} text labels") + print(f" - {len(structure['lines'])} horizontal lines") + print(f" - {len(structure['checkboxes'])} checkboxes") + print(f" - {len(structure['row_boundaries'])} row boundaries") + print(f"Saved to {output_path}") + + +if __name__ == "__main__": + main() diff --git a/dot_agents/skills/pdf/scripts/extract_form_field_info.py b/dot_agents/skills/pdf/scripts/extract_form_field_info.py new file mode 100644 index 00000000..7f69dd4a --- /dev/null +++ b/dot_agents/skills/pdf/scripts/extract_form_field_info.py @@ -0,0 +1,130 @@ +import json +import sys + +from pypdf import PdfReader + + +def get_full_annotation_field_id(annotation): + components = [] + while annotation: + field_name = annotation.get("/T") + if field_name: + components.append(field_name) + annotation = annotation.get("/Parent") + return ".".join(reversed(components)) if components else None + + +def make_field_dict(field, field_id): + field_dict = {"field_id": field_id} + ft = field.get("/FT") + if ft == "/Tx": + field_dict["type"] = "text" + elif ft == "/Btn": + field_dict["type"] = "checkbox" + states = field.get("/_States_", []) + if len(states) == 2: + if "/Off" in states: + field_dict["checked_value"] = ( + states[0] if states[0] != "/Off" else states[1] + ) + field_dict["unchecked_value"] = "/Off" + else: + print( + f"Unexpected state values for checkbox `${field_id}`. Its checked and unchecked values may not be correct; if you're trying to check it, visually verify the results." + ) + field_dict["checked_value"] = states[0] + field_dict["unchecked_value"] = states[1] + elif ft == "/Ch": + field_dict["type"] = "choice" + states = field.get("/_States_", []) + field_dict["choice_options"] = [ + { + "value": state[0], + "text": state[1], + } + for state in states + ] + else: + field_dict["type"] = f"unknown ({ft})" + return field_dict + + +def get_field_info(reader: PdfReader): + fields = reader.get_fields() + + field_info_by_id = {} + possible_radio_names = set() + + for field_id, field in fields.items(): + if field.get("/Kids"): + if field.get("/FT") == "/Btn": + possible_radio_names.add(field_id) + continue + field_info_by_id[field_id] = make_field_dict(field, field_id) + + radio_fields_by_id = {} + + for page_index, page in enumerate(reader.pages): + annotations = page.get("/Annots", []) + for ann in annotations: + field_id = get_full_annotation_field_id(ann) + if field_id in field_info_by_id: + field_info_by_id[field_id]["page"] = page_index + 1 + field_info_by_id[field_id]["rect"] = ann.get("/Rect") + elif field_id in possible_radio_names: + try: + on_values = [v for v in ann["/AP"]["/N"] if v != "/Off"] + except KeyError: + continue + if len(on_values) == 1: + rect = ann.get("/Rect") + if field_id not in radio_fields_by_id: + radio_fields_by_id[field_id] = { + "field_id": field_id, + "type": "radio_group", + "page": page_index + 1, + "radio_options": [], + } + radio_fields_by_id[field_id]["radio_options"].append( + { + "value": on_values[0], + "rect": rect, + } + ) + + fields_with_location = [] + for field_info in field_info_by_id.values(): + if "page" in field_info: + fields_with_location.append(field_info) + else: + print( + f"Unable to determine location for field id: {field_info.get('field_id')}, ignoring" + ) + + def sort_key(f): + if "radio_options" in f: + rect = f["radio_options"][0]["rect"] or [0, 0, 0, 0] + else: + rect = f.get("rect") or [0, 0, 0, 0] + adjusted_position = [-rect[1], rect[0]] + return [f.get("page"), adjusted_position] + + sorted_fields = fields_with_location + list(radio_fields_by_id.values()) + sorted_fields.sort(key=sort_key) + + return sorted_fields + + +def write_field_info(pdf_path: str, json_output_path: str): + reader = PdfReader(pdf_path) + field_info = get_field_info(reader) + with open(json_output_path, "w") as f: + json.dump(field_info, f, indent=2) + print(f"Wrote {len(field_info)} fields to {json_output_path}") + + +if __name__ == "__main__": + if len(sys.argv) != 3: + print("Usage: extract_form_field_info.py [input pdf] [output json]") + sys.exit(1) + write_field_info(sys.argv[1], sys.argv[2]) diff --git a/dot_agents/skills/pdf/scripts/fill_fillable_fields.py b/dot_agents/skills/pdf/scripts/fill_fillable_fields.py new file mode 100644 index 00000000..4181689b --- /dev/null +++ b/dot_agents/skills/pdf/scripts/fill_fillable_fields.py @@ -0,0 +1,104 @@ +import json +import sys + +from pypdf import PdfReader, PdfWriter + +from extract_form_field_info import get_field_info + + +def fill_pdf_fields(input_pdf_path: str, fields_json_path: str, output_pdf_path: str): + with open(fields_json_path) as f: + fields = json.load(f) + fields_by_page = {} + for field in fields: + if "value" in field: + field_id = field["field_id"] + page = field["page"] + if page not in fields_by_page: + fields_by_page[page] = {} + fields_by_page[page][field_id] = field["value"] + + reader = PdfReader(input_pdf_path) + + has_error = False + field_info = get_field_info(reader) + fields_by_ids = {f["field_id"]: f for f in field_info} + for field in fields: + existing_field = fields_by_ids.get(field["field_id"]) + if not existing_field: + has_error = True + print(f"ERROR: `{field['field_id']}` is not a valid field ID") + elif field["page"] != existing_field["page"]: + has_error = True + print( + f"ERROR: Incorrect page number for `{field['field_id']}` (got {field['page']}, expected {existing_field['page']})" + ) + else: + if "value" in field: + err = validation_error_for_field_value(existing_field, field["value"]) + if err: + print(err) + has_error = True + if has_error: + sys.exit(1) + + writer = PdfWriter(clone_from=reader) + for page, field_values in fields_by_page.items(): + writer.update_page_form_field_values( + writer.pages[page - 1], field_values, auto_regenerate=False + ) + + writer.set_need_appearances_writer(True) + + with open(output_pdf_path, "wb") as f: + writer.write(f) + + +def validation_error_for_field_value(field_info, field_value): + field_type = field_info["type"] + field_id = field_info["field_id"] + if field_type == "checkbox": + checked_val = field_info["checked_value"] + unchecked_val = field_info["unchecked_value"] + if field_value != checked_val and field_value != unchecked_val: + return f'ERROR: Invalid value "{field_value}" for checkbox field "{field_id}". The checked value is "{checked_val}" and the unchecked value is "{unchecked_val}"' + elif field_type == "radio_group": + option_values = [opt["value"] for opt in field_info["radio_options"]] + if field_value not in option_values: + return f'ERROR: Invalid value "{field_value}" for radio group field "{field_id}". Valid values are: {option_values}' + elif field_type == "choice": + choice_values = [opt["value"] for opt in field_info["choice_options"]] + if field_value not in choice_values: + return f'ERROR: Invalid value "{field_value}" for choice field "{field_id}". Valid values are: {choice_values}' + return None + + +def monkeypatch_pydpf_method(): + from pypdf.generic import DictionaryObject + from pypdf.constants import FieldDictionaryAttributes + + original_get_inherited = DictionaryObject.get_inherited + + def patched_get_inherited(self, key: str, default=None): + result = original_get_inherited(self, key, default) + if key == FieldDictionaryAttributes.Opt: + if isinstance(result, list) and all( + isinstance(v, list) and len(v) == 2 for v in result + ): + result = [r[0] for r in result] + return result + + DictionaryObject.get_inherited = patched_get_inherited + + +if __name__ == "__main__": + if len(sys.argv) != 4: + print( + "Usage: fill_fillable_fields.py [input pdf] [field_values.json] [output pdf]" + ) + sys.exit(1) + monkeypatch_pydpf_method() + input_pdf = sys.argv[1] + fields_json = sys.argv[2] + output_pdf = sys.argv[3] + fill_pdf_fields(input_pdf, fields_json, output_pdf) diff --git a/dot_agents/skills/pdf/scripts/fill_pdf_form_with_annotations.py b/dot_agents/skills/pdf/scripts/fill_pdf_form_with_annotations.py new file mode 100644 index 00000000..46f6df30 --- /dev/null +++ b/dot_agents/skills/pdf/scripts/fill_pdf_form_with_annotations.py @@ -0,0 +1,110 @@ +import json +import sys + +from pypdf import PdfReader, PdfWriter +from pypdf.annotations import FreeText + + +def transform_from_image_coords(bbox, image_width, image_height, pdf_width, pdf_height): + x_scale = pdf_width / image_width + y_scale = pdf_height / image_height + + left = bbox[0] * x_scale + right = bbox[2] * x_scale + + top = pdf_height - (bbox[1] * y_scale) + bottom = pdf_height - (bbox[3] * y_scale) + + return left, bottom, right, top + + +def transform_from_pdf_coords(bbox, pdf_height): + left = bbox[0] + right = bbox[2] + + pypdf_top = pdf_height - bbox[1] + pypdf_bottom = pdf_height - bbox[3] + + return left, pypdf_bottom, right, pypdf_top + + +def fill_pdf_form(input_pdf_path, fields_json_path, output_pdf_path): + + with open(fields_json_path, "r") as f: + fields_data = json.load(f) + + reader = PdfReader(input_pdf_path) + writer = PdfWriter() + + writer.append(reader) + + pdf_dimensions = {} + for i, page in enumerate(reader.pages): + mediabox = page.mediabox + pdf_dimensions[i + 1] = [mediabox.width, mediabox.height] + + annotations = [] + for field in fields_data["form_fields"]: + page_num = field["page_number"] + + page_info = next( + p for p in fields_data["pages"] if p["page_number"] == page_num + ) + pdf_width, pdf_height = pdf_dimensions[page_num] + + if "pdf_width" in page_info: + transformed_entry_box = transform_from_pdf_coords( + field["entry_bounding_box"], float(pdf_height) + ) + else: + image_width = page_info["image_width"] + image_height = page_info["image_height"] + transformed_entry_box = transform_from_image_coords( + field["entry_bounding_box"], + image_width, + image_height, + float(pdf_width), + float(pdf_height), + ) + + if "entry_text" not in field or "text" not in field["entry_text"]: + continue + entry_text = field["entry_text"] + text = entry_text["text"] + if not text: + continue + + font_name = entry_text.get("font", "Arial") + font_size = str(entry_text.get("font_size", 14)) + "pt" + font_color = entry_text.get("font_color", "000000") + + annotation = FreeText( + text=text, + rect=transformed_entry_box, + font=font_name, + font_size=font_size, + font_color=font_color, + border_color=None, + background_color=None, + ) + annotations.append(annotation) + writer.add_annotation(page_number=page_num - 1, annotation=annotation) + + with open(output_pdf_path, "wb") as output: + writer.write(output) + + print(f"Successfully filled PDF form and saved to {output_pdf_path}") + print(f"Added {len(annotations)} text annotations") + + +if __name__ == "__main__": + if len(sys.argv) != 4: + print( + "Usage: fill_pdf_form_with_annotations.py [input pdf] [fields.json] [output pdf]" + ) + sys.exit(1) + input_pdf = sys.argv[1] + fields_json = sys.argv[2] + output_pdf = sys.argv[3] + + fill_pdf_form(input_pdf, fields_json, output_pdf) diff --git a/dot_agents/skills/pdf/scripts/literal_create_validation_image.py b/dot_agents/skills/pdf/scripts/literal_create_validation_image.py new file mode 100644 index 00000000..d14fd870 --- /dev/null +++ b/dot_agents/skills/pdf/scripts/literal_create_validation_image.py @@ -0,0 +1,41 @@ +import json +import sys + +from PIL import Image, ImageDraw + + +def create_validation_image(page_number, fields_json_path, input_path, output_path): + with open(fields_json_path, "r") as f: + data = json.load(f) + + img = Image.open(input_path) + draw = ImageDraw.Draw(img) + num_boxes = 0 + + for field in data["form_fields"]: + if field["page_number"] == page_number: + entry_box = field["entry_bounding_box"] + label_box = field["label_bounding_box"] + draw.rectangle(entry_box, outline="red", width=2) + draw.rectangle(label_box, outline="blue", width=2) + num_boxes += 2 + + img.save(output_path) + print( + f"Created validation image at {output_path} with {num_boxes} bounding boxes" + ) + + +if __name__ == "__main__": + if len(sys.argv) != 5: + print( + "Usage: create_validation_image.py [page number] [fields.json file] [input image path] [output image path]" + ) + sys.exit(1) + page_number = int(sys.argv[1]) + fields_json_path = sys.argv[2] + input_image_path = sys.argv[3] + output_image_path = sys.argv[4] + create_validation_image( + page_number, fields_json_path, input_image_path, output_image_path + ) diff --git a/dot_agents/skills/simplify/SKILL.md b/dot_agents/skills/simplify/SKILL.md index ee5fab9f..0ceb2fea 100644 --- a/dot_agents/skills/simplify/SKILL.md +++ b/dot_agents/skills/simplify/SKILL.md @@ -1,5 +1,4 @@ --- -disable-model-invocation: true name: simplify description: Use when the user wants recently modified code simplified without changing behavior. Apply the repository's current standards, improve clarity, remove unnecessary complexity, and keep the work scoped to code touched in the current session unless the user asks for a broader pass. --- diff --git a/dot_agents/skills/web-search/executable_search.sh b/dot_agents/skills/web-search/executable_search.sh old mode 100755 new mode 100644 diff --git a/dot_agents/skills/web-search/literal_executable_search.sh b/dot_agents/skills/web-search/literal_executable_search.sh new file mode 100644 index 00000000..d7ae6e21 --- /dev/null +++ b/dot_agents/skills/web-search/literal_executable_search.sh @@ -0,0 +1,94 @@ +#!/usr/bin/env bash +# kagi-only web research dispatcher. +# Modes: search | quick | ask | summarize. +# Every mode post-processes kagi-cli JSON down to the minimal meaningful text +# so the caller's context never sees HTML, favicons, traces, or metadata. +set -uo pipefail + +MODE="" +INPUT="" +LIMIT=5 +THREAD_ID="" +SUMMARY_TYPE="summary" +LENGTH="" +FOLLOWUPS=0 + +usage() { + cat <<'EOF' +Usage: + search.sh "" [flags] + +Modes: + search List results (title + url + snippet). Cheapest, no synthesis. + quick Grounded answer with ranked source links. Default for facts. + ask Kagi Assistant for deeper synthesis / multi-step reasoning. + summarize Condense one URL into key text. + +Flags: + --limit search: number of results (default 5) + --thread-id ask: continue an existing assistant thread + --summary-type summarize: summary | keypoints (default summary) + --length summarize: short | digest | etc. + --followups quick: also print follow-up questions + +Examples: + search.sh quick "latest stable rust version" + search.sh search "vite 7 breaking changes" --limit 8 + search.sh ask "compare uv vs poetry for monorepos" + search.sh summarize "https://example.com/article" +EOF +} + +[[ $# -eq 0 ]] && { usage; exit 2; } +MODE="$1"; shift + +while [[ $# -gt 0 ]]; do + case "$1" in + --limit) LIMIT="$2"; shift 2 ;; + --thread-id) THREAD_ID="$2"; shift 2 ;; + --summary-type) SUMMARY_TYPE="$2"; shift 2 ;; + --length) LENGTH="$2"; shift 2 ;; + --followups) FOLLOWUPS=1; shift ;; + -h|--help) usage; exit 0 ;; + -*) echo "Unknown flag: $1" >&2; usage; exit 2 ;; + *) INPUT="${INPUT:+$INPUT }$1"; shift ;; + esac +done + +[[ -z "$INPUT" ]] && { usage; exit 2; } + +command -v kagi >/dev/null 2>&1 || { echo "ERROR: kagi CLI not on PATH" >&2; exit 127; } +# cd $HOME so kagi-cli resolves the session token in ~/.kagi.toml +cd "$HOME" || exit 1 + +case "$MODE" in + search) + kagi search --limit "$LIMIT" "$INPUT" \ + | jq -r '.data[]? | select(.url) | "- [\(.title)](\(.url))\n \(.snippet // "" | gsub("\\s+"; " "))"' + ;; + + quick) + out="$(kagi quick "$INPUT")" + jq -r '.message.markdown' <<<"$out" + echo + jq -r '.references.markdown // empty' <<<"$out" + [[ "$FOLLOWUPS" -eq 1 ]] && jq -r '(.followup_questions // [])[] | "- " + .' <<<"$out" + ;; + + ask) + if [[ -n "$THREAD_ID" ]]; then + kagi assistant --thread-id "$THREAD_ID" --format markdown "$INPUT" + else + kagi assistant --format markdown "$INPUT" + fi + ;; + + summarize) + args=(summarize --subscriber --url "$INPUT" --summary-type "$SUMMARY_TYPE") + [[ -n "$LENGTH" ]] && args+=(--length "$LENGTH") + kagi "${args[@]}" | jq -r '.data.markdown // .data.output // empty' + ;; + + *) + echo "Unknown mode: $MODE" >&2; usage; exit 2 ;; +esac diff --git a/dot_config/private_fish/functions/skills.fish b/dot_config/private_fish/functions/skills.fish deleted file mode 100644 index 97595ee6..00000000 --- a/dot_config/private_fish/functions/skills.fish +++ /dev/null @@ -1,9 +0,0 @@ -function skills --description 'skills CLI with invocation-policy enforcement after updates' - command skills $argv - set -l status_saved $status - switch "$argv[1]" - case update upgrade add - skills-invocation - end - return $status_saved -end diff --git a/dot_pi/agent/AGENTS.md b/dot_pi/agent/AGENTS.md index a685da5e..3e599ce9 100644 --- a/dot_pi/agent/AGENTS.md +++ b/dot_pi/agent/AGENTS.md @@ -11,19 +11,3 @@ - Always use `fd` instead of `find`. - Use `jq` for JSON processing - Fish shell is the primary shell -- Clone new repos with `ghq get ` (root is `~/Code`, layout `~/Code///`) - -## Coding fundamental principles - -- Always have a bird's eye view of the code - "See the forest, not just the trees" -- Never do backwards compatibility, we move only forward -- Do not have any dead code, unused code should be cleaned up -- Do not write any comments, code is truth -- Do not store what you can compute. Do not send what can be derived. Each piece of data exists in one please -- Paul Dirac's beauty in code -- Duplication is decay, decay is corruption, corruption is death -- Expose what must be exposed, hide what must be hidden -- Every line will be judged at the scales -- Occam's Razor in code - -@RTK.md diff --git a/private_dot_local/bin/executable_skills-invocation b/private_dot_local/bin/executable_skills-invocation deleted file mode 100644 index 8cbe58c2..00000000 --- a/private_dot_local/bin/executable_skills-invocation +++ /dev/null @@ -1,55 +0,0 @@ -#!/bin/bash -set -euo pipefail - -# Enforces skill invocation policy on github-sourced skills. -# -# `skills update` re-clones github-sourced skills, wiping any local edits to -# their SKILL.md. Local-sourced skills (chezmoi, local repos) keep their edits -# and define their own policy directly in their frontmatter. -# -# This script is the single source of truth for which github-sourced skills the -# model may auto-invoke. Everything else gets `disable-model-invocation: true`, -# making it reachable only via `/skill:`. - -LOCK_FILE="$HOME/.agents/.skill-lock.json" -SKILLS_DIR="$HOME/.agents/skills" - -# Skills the model is allowed to auto-invoke. -AUTOLOAD=(ai-search humanizer kagi-cli web-search pdf summarize) - -is_autoload() { - local name="$1" - for a in "${AUTOLOAD[@]}"; do - [[ "$a" == "$name" ]] && return 0 - done - return 1 -} - -set_disable() { - local file="$1" - grep -q '^disable-model-invocation:' "$file" && return 0 - awk 'NR==1{print; print "disable-model-invocation: true"; next} {print}' \ - "$file" >"$file.tmp" && mv "$file.tmp" "$file" -} - -clear_disable() { - local file="$1" - grep -q '^disable-model-invocation:' "$file" || return 0 - grep -v '^disable-model-invocation:' "$file" >"$file.tmp" && mv "$file.tmp" "$file" -} - -[[ -f "$LOCK_FILE" ]] || { - echo "No lock file at $LOCK_FILE" >&2 - exit 1 -} - -jq -r '.skills | to_entries[] | select(.value.sourceType == "github") | .key' "$LOCK_FILE" | - while read -r name; do - file="$SKILLS_DIR/$name/SKILL.md" - [[ -f "$file" ]] || continue - if is_autoload "$name"; then - clear_disable "$file" - else - set_disable "$file" - fi - done