Install any skill in seconds. Free to start, no credit card required.
Get Started Free →Word (.docx/.doc) 文档全量解析。覆盖:正文/段落文本提取、表格数据提取、高亮/颜色格式读取、多文件汇总对比、嵌入图片转 caption。
.claude/skills/opensensenova-word-analysis/SKILL.md| Test case | Without → With | Effect | Δ tokens | Δ turns |
|---|---|---|---|---|
| case-25 | ✗→✓ | ▲ Improved | 31% | 0% |
| case-06 | ✗→✓ | ▲ Improved | 128% | 0% |
| case-14 | ✗→✓ | ▲ Improved | 76% | 0% |
| case-15 | ✗→✓ | ▲ Improved | 60% | 0% |
| case-16 | ✗→✓ | ▲ Improved | 175% | 0% |
pythonfrom docx import Document import os # python-docx is available; for .doc (old format) convert via libreoffice first def load_doc(path): """Load .docx directly; convert .doc to .docx first if needed.""" if path.lower().endswith('.doc'): import subprocess out_dir = os.path.dirname(path) subprocess.run( ['libreoffice', '--headless', '--convert-to', 'docx', '--outdir', out_dir, path], check=True, capture_output=True ) path = path.rsplit('.', 1)[0] + '.docx' return Document(path)
pythondef extract_full_text(doc_path): """Extract all text: paragraphs + table cells, in document order.""" doc = load_doc(doc_path) lines = [] # Iterate paragraphs and tables in body order from docx.oxml.ns import qn for block in doc.element.body: tag = block.tag.split('}')[-1] if tag == 'p': # Paragraph from docx.text.paragraph import Paragraph para = Paragraph(block, doc) text = para.text.strip() if text: lines.append(text) elif tag == 'tbl': # Table from docx.table import Table tbl = Table(block, doc) for row in tbl.rows: row_text = '\t'.join(cell.text.strip() for cell in row.cells) if row_text.strip(): lines.append(row_text) return '\n'.join(lines) # Usage text = extract_full_text("/mnt/data/doc.docx") print(text[:2000]) # preview first 2000 chars
pythonimport pandas as pd def extract_all_tables(doc_path): """Extract all tables from a Word document as list of DataFrames.""" doc = load_doc(doc_path) tables = [] for i, tbl in enumerate(doc.tables): rows = [] for row in tbl.rows: rows.append([cell.text.strip() for cell in row.cells]) if not rows: continue # Use first row as header if it looks like a header df = pd.DataFrame(rows[1:], columns=rows[0]) if rows else pd.DataFrame() tables.append((i, df)) print(f"Table {i}: {df.shape[0]} rows × {df.shape[1]} cols") print(df.head(3)) return tables # Usage tables = extract_all_tables("/mnt/data/doc.docx")
Some questions require reading cell background color or text highlight color (e.g., "标黄的行", "红色文字"). Use XML-level access:
pythonfrom docx import Document from docx.oxml.ns import qn from lxml import etree def get_paragraph_highlight(para): """Return highlight color name of first run, or None.""" for run in para.runs: rPr = run._r.find(qn('w:rPr')) if rPr is not None: hl = rPr.find(qn('w:highlight')) if hl is not None: return hl.get(qn('w:val')) # e.g. 'yellow', 'cyan', 'red' return None def get_table_cell_shading(cell): """Return background color hex of a table cell, or None.""" tcPr = cell._tc.find(qn('w:tcPr')) if tcPr is not None: shd = tcPr.find(qn('w:shd')) if shd is not None: return shd.get(qn('w:fill')) # hex color, e.g. 'FFFF00' return None # Example: find all highlighted paragraphs def find_highlighted_rows(doc_path, color='yellow'): doc = load_doc(doc_path) highlighted = [] for i, para in enumerate(doc.paragraphs): hl = get_paragraph_highlight(para) if hl == color or (color == 'yellow' and hl in ('yellow', 'FFFF00')): highlighted.append((i, para.text)) return highlighted # For table cells with yellow background: def find_highlighted_table_cells(doc_path, fill_colors=('FFFF00', 'FFD700')): doc = load_doc(doc_path) results = [] for t_idx, tbl in enumerate(doc.tables): for r_idx, row in enumerate(tbl.rows): for c_idx, cell in enumerate(row.cells): color = get_table_cell_shading(cell) if color and color.upper() in fill_colors: results.append({ 'table': t_idx, 'row': r_idx, 'col': c_idx, 'color': color, 'text': cell.text.strip() }) return results
When the user asks about "these files" or the input is a directory:
pythondef process_all_docs(file_list, extractor_fn): """Apply extractor to all files and aggregate results.""" all_results = [] for path in file_list: print(f"\n=== Processing: {os.path.basename(path)} ===") try: result = extractor_fn(path) all_results.append({'file': os.path.basename(path), 'data': result}) except Exception as e: print(f" ERROR: {e}") return all_results # Example: extract text from all .docx in a directory doc_files = [f for f in all_files if f.lower().endswith(('.docx', '.doc'))] results = process_all_docs(doc_files, extract_full_text)
When a Word doc contains embedded images (charts, screenshots):
pythonimport zipfile, io, subprocess, json CAPTION = "/path/to/skills/sn-da-image-caption/scripts/caption.py" def extract_and_caption_images(doc_path, prompt=None): """Extract all images from .docx and caption each one.""" # .docx is a ZIP archive; images are in word/media/ results = [] with zipfile.ZipFile(doc_path, 'r') as z: media_files = [n for n in z.namelist() if n.startswith('word/media/')] for media in media_files: ext = os.path.splitext(media)[-1].lower() if ext not in ('.png', '.jpg', '.jpeg', '.gif', '.bmp', '.wmf', '.emf'): continue # Save to temp tmp_path = f"/tmp/{os.path.basename(media)}" with z.open(media) as src, open(tmp_path, 'wb') as dst: dst.write(src.read()) # Caption cmd = ["python3", CAPTION, tmp_path, "--json"] if prompt: cmd += ["--prompt", prompt] r = subprocess.run(cmd, capture_output=True, text=True, timeout=60) if r.returncode == 0: desc = json.loads(r.stdout).get("description", "") results.append({'image': media, 'caption': desc}) print(f" {media}: {desc[:100]}...") else: print(f" {media}: caption failed — {r.stderr[:80]}") return results
pythonfrom docx.shared import Pt def check_font_sizes(doc_path): doc = load_doc(doc_path) issues = [] for i, para in enumerate(doc.paragraphs): for run in para.runs: size = run.font.size size_pt = size.pt if size else None # Also check style-level font if size_pt is None: style_size = run.style.font.size if run.style else None size_pt = style_size.pt if style_size else None issues.append({'para': i, 'text': run.text[:30], 'size_pt': size_pt}) return issues
pythondef find_keyword(doc_path, keyword): text = extract_full_text(doc_path) idx = text.find(keyword) if idx >= 0: context = text[max(0, idx-100):idx+200] print(f"Found '{keyword}' at pos {idx}:\n{context}") else: print(f"'{keyword}' not found. Try broader search.") # Try case-insensitive or partial match for kw in keyword.split(): if kw in text: print(f" Partial match for '{kw}'")
| Pitfall | Fix | |---------|-----| | Only read doc.paragraphs, miss tables | Use the body-order iterator in Method 1 | | Single file when input is multi-file | Check os.path.isdir(), iterate all | | Highlighted cells not detected | Use XML-level w:shd / w:highlight (Method 3) | | .doc format fails to open | Convert to .docx via libreoffice (Method 0) | | Embedded charts look empty | Extract images from ZIP, caption each (Method 5) | | Font size is None | Check both run-level and style-level (Method for font check) |
| Case | Status | Duration (ms) | Turns | Tokens | Tool calls | ||||||||
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| Without | With | Δ | Without | With | Δ | Without | With | Δ | Without | With | Δ | ||
case-08 | pass→pass | 13,384 | 11,320 | -15% | 1 | 1 | 0% | 2,528 | 4,794 | +90% | 0 | 0 | — |
case-01 | fail→fail | 5,916 | 5,860 | -1% | 1 | 1 | 0% | 281 | 2,813 | +901% | 0 | 0 | — |
case-02 | pass→pass | 11,187 | 7,978 | -29% | 1 | 1 | 0% | 2,264 | 4,333 | +91% | 0 | 0 | — |
case-25 | fail→pass | 26,545 | 24,038 | -9% | 1 | 1 | 0% | 6,164 | 8,080 | +31% | 0 | 0 | — |
case-03 | pass→pass | 6,843 | 4,924 | -28% | 1 | 1 | 0% | 1,418 | 3,427 | +142% | 0 | 0 | — |
case-04 | pass→pass | 8,258 | 6,425 | -22% | 1 | 1 | 0% | 1,663 | 3,890 | +134% | 0 | 0 | — |
case-05 | pass→pass | 10,169 | 8,762 | -14% | 1 | 1 | 0% | 2,115 | 4,284 | +103% | 0 | 0 | — |
case-06 | fail→pass | 9,512 | 8,589 | -10% | 1 | 1 | 0% | 1,932 | 4,398 | +128% | 0 | 0 | — |
case-07 | pass→pass | 12,474 | 13,134 | +5% | 1 | 1 | 0% | 1,988 | 5,461 | +175% | 0 | 0 | — |
case-09 | pass→pass | 14,454 | 26,704 | +85% | 1 | 1 | 0% | 3,149 | 7,300 | +132% | 0 | 0 | — |
case-10 | pass→pass | 9,007 | 7,278 | -19% | 1 | 1 | 0% | 1,839 | 4,009 | +118% | 0 | 0 | — |
case-11 | pass→pass | 9,309 | 6,364 | -32% | 1 | 1 | 0% | 2,056 | 3,966 | +93% | 0 | 0 | — |
case-12 | pass→pass | 12,362 | 9,736 | -21% | 1 | 1 | 0% | 2,519 | 4,552 | +81% | 0 | 0 | — |
case-13 | pass→pass | 6,642 | 3,687 | -44% | 1 | 1 | 0% | 1,334 | 3,282 | +146% | 0 | 0 | — |
case-14 | fail→pass | 9,857 | 4,628 | -53% | 1 | 1 | 0% | 1,939 | 3,411 | +76% | 0 | 0 | — |
case-15 | fail→pass | 11,396 | 6,294 | -45% | 1 | 1 | 0% | 2,467 | 3,953 | +60% | 0 | 0 | — |
case-16 | fail→pass | 20,981 | 6,711 | -68% | 1 | 1 | 0% | 1,437 | 3,954 | +175% | 0 | 0 | — |
case-17 | pass→pass | 8,748 | 5,150 | -41% | 1 | 1 | 0% | 1,647 | 3,409 | +107% | 0 | 0 | — |
case-18 | pass→pass | 8,238 | 4,469 | -46% | 1 | 1 | 0% | 1,476 | 3,265 | +121% | 0 | 0 | — |
case-19 | pass→pass | 10,076 | 5,866 | -42% | 1 | 1 | 0% | 2,093 | 3,791 | +81% | 0 | 0 | — |
case-20 | fail→fail | 6,383 | 8,781 | +38% | 1 | 1 | 0% | 120 | 2,848 | +2273% | 0 | 0 | — |
case-21 | pass→pass | 10,154 | 10,107 | -0% | 1 | 1 | 0% | 1,960 | 4,374 | +123% | 0 | 0 | — |
case-22 | fail→pass | 16,392 | 7,634 | -53% | 1 | 1 | 0% | 3,432 | 4,296 | +25% | 0 | 0 | — |
case-23 | pass→pass | 6,190 | 4,432 | -28% | 1 | 1 | 0% | 1,191 | 3,440 | +189% | 0 | 0 | — |
case-24 | pass→fail | 27,773 | 5,735 | -79% | 1 | 1 | 0% | 6,175 | 2,705 | -56% | 0 | 0 | — |
DecimalAI ran this skill against gemini-3.6-flash twice over the same eval suite — once with the skill loaded and once without — and compared the two runs case by case. 25 cases were attempted, and 21 counted toward the lift figure. The other 4 produced results that are not comparable between the two arms, so they are excluded from the headline rather than averaged into it. The headline lift of +20 percentage points is the difference between those two pass rates over the 21 comparable cases. 1 case got worse with the skill loaded, and it is included in that figure.
Without the skill loaded, the model failed this case. With it loaded, the same prompt on the same model passed. This is one improved case from the latest verified run; every case, including any that regressed, is in the table above.
Other measured skills in the registry, with their headline benchmark lift.