Install any skill in seconds. Free to start, no credit card required.
Get Started Free →PPT (.pptx/.ppt) 全量解析。覆盖:所有 slide 文本/表格/图表提取、嵌入图片 caption、纯图片 slide 渲染识别、数据标签提取。
.claude/skills/opensensenova-ppt-analysis/SKILL.md| Test case | Without → With | Effect | Δ tokens | Δ turns |
|---|---|---|---|---|
| case-03 | ✗→✓ | ▲ Improved | 848% | 0% |
| case-06 | ✗→✓ | ▲ Improved | 1145% | 0% |
| case-08 | ✗→✓ | ▲ Improved | 106% | 0% |
| case-10 | ✗→✓ | ▲ Improved | 134% | 0% |
| case-11 | ✗→✓ | ▲ Improved | 123% | 0% |
pythonfrom pptx import Presentation from pptx.util import Inches import os, subprocess, json # python-pptx is available # For .ppt (old binary format): convert via libreoffice def load_pptx(path): if path.lower().endswith('.ppt'): import subprocess out_dir = os.path.dirname(path) subprocess.run( ['libreoffice', '--headless', '--convert-to', 'pptx', '--outdir', out_dir, path], check=True, capture_output=True ) path = path.rsplit('.', 1)[0] + '.pptx' return Presentation(path), path
pythondef extract_all_slides_text(pptx_path): """ Extract text from every slide: text frames, tables, chart titles. For slides with no extractable text, flag them for image captioning. """ prs, _ = load_pptx(pptx_path) slides_data = [] for slide_num, slide in enumerate(prs.slides, start=1): slide_texts = [] has_text = False for shape in slide.shapes: # Text frame (most common) if shape.has_text_frame: for para in shape.text_frame.paragraphs: text = para.text.strip() if text: slide_texts.append(text) has_text = True # Table if shape.has_table: tbl = shape.table for row in tbl.rows: row_text = '\t'.join(cell.text.strip() for cell in row.cells) if row_text.strip(): slide_texts.append(row_text) has_text = True # Chart title if shape.shape_type == 3: # MSO_SHAPE_TYPE.CHART try: if shape.chart.has_title: title = shape.chart.chart_title.text_frame.text slide_texts.append(f"[Chart: {title}]") has_text = True except Exception: pass slides_data.append({ 'slide': slide_num, 'text': '\n'.join(slide_texts), 'has_text': has_text, 'needs_caption': not has_text # flag image-only slides }) print(f"Total slides: {len(slides_data)}") image_only = sum(1 for s in slides_data if s['needs_caption']) print(f"Slides with text: {len(slides_data) - image_only}, image-only: {image_only}") return slides_data
pythonimport pandas as pd def extract_pptx_tables(pptx_path): """Extract all tables from all slides as DataFrames.""" prs, _ = load_pptx(pptx_path) all_tables = [] for slide_num, slide in enumerate(prs.slides, start=1): for shape in slide.shapes: if not shape.has_table: continue tbl = shape.table rows = [] for row in tbl.rows: rows.append([cell.text.strip() for cell in row.cells]) if not rows: continue # Use first row as header try: df = pd.DataFrame(rows[1:], columns=rows[0]) except Exception: df = pd.DataFrame(rows) all_tables.append({'slide': slide_num, 'df': df}) print(f" Slide {slide_num}: table {df.shape[0]}r × {df.shape[1]}c") print(df.head(3).to_string()) return all_tables
python-pptx can read Chart data when it's stored as embedded Excel data. If that fails, fall back to captioning the slide image.
pythondef extract_chart_data(pptx_path): """ Extract data series from Chart shapes. Returns list of {slide, chart_title, series_name, categories, values}. """ prs, _ = load_pptx(pptx_path) charts = [] for slide_num, slide in enumerate(prs.slides, start=1): for shape in slide.shapes: if shape.shape_type != 3: # not a chart continue try: chart = shape.chart title = chart.chart_title.text_frame.text if chart.has_title else f"Chart_S{slide_num}" for plot in chart.plots: for series in plot.series: try: categories = [str(pt.label) for pt in series.data_labels] if hasattr(series, 'data_labels') else [] values = [pt.value for pt in series.values] if hasattr(series, 'values') else [] # Alternative: use xChart data if not values: values = list(series.values) except Exception as e: values = [] categories = [] charts.append({ 'slide': slide_num, 'chart_title': title, 'series': getattr(series, 'name', ''), 'categories': categories, 'values': values }) except Exception as e: print(f" Slide {slide_num}: chart extraction failed ({e}) — will use caption") return charts
When a slide has no extractable text (pure image/screenshot slides):
pythonimport fitz # PyMuPDF can also render PPTX via LibreOffice conversion CAPTION = "/path/to/skills/sn-da-image-caption/scripts/caption.py" def caption_image_slides(pptx_path, slides_data, prompt=None): """ For slides flagged as 'needs_caption', render to PNG and caption. Uses LibreOffice to convert PPTX to PDF first, then renders pages. """ image_slides = [s for s in slides_data if s['needs_caption']] if not image_slides: print("No image-only slides to caption.") return slides_data # Convert PPTX → PDF (preserves slide visuals) out_dir = "/tmp" r = subprocess.run( ['libreoffice', '--headless', '--convert-to', 'pdf', '--outdir', out_dir, pptx_path], capture_output=True, text=True ) pdf_name = os.path.basename(pptx_path).rsplit('.', 1)[0] + '.pdf' pdf_path = os.path.join(out_dir, pdf_name) if not os.path.exists(pdf_path): print(f"LibreOffice conversion failed: {r.stderr[:200]}") return slides_data # Render each image-only slide doc = fitz.open(pdf_path) for s in image_slides: page_idx = s['slide'] - 1 # 0-indexed if page_idx >= len(doc): continue page = doc[page_idx] mat = fitz.Matrix(150/72, 150/72) pix = page.get_pixmap(matrix=mat) img_path = f"/tmp/slide_{s['slide']}.png" pix.save(img_path) # Caption the slide image cmd = ["python3", CAPTION, img_path, "--json"] p = prompt or "提取幻灯片中所有文字、数值和表格内容,保持结构,Markdown格式输出。" cmd += ["--prompt", p] cr = subprocess.run(cmd, capture_output=True, text=True, timeout=90) if cr.returncode == 0: desc = json.loads(cr.stdout).get("description", "") s['text'] = desc s['needs_caption'] = False print(f" Slide {s['slide']}: captioned ({len(desc)} chars)") else: print(f" Slide {s['slide']}: caption failed — {cr.stderr[:80]}") doc.close() return slides_data
pythondef find_in_pptx(pptx_path, keyword, slides_data=None): """Find keyword across all slides (after text extraction + captioning).""" if slides_data is None: slides_data = extract_all_slides_text(pptx_path) results = [] for s in slides_data: if keyword in s.get('text', ''): idx = s['text'].find(keyword) context = s['text'][max(0, idx-100):idx+200] results.append({'slide': s['slide'], 'context': context}) print(f"'{keyword}' found in {len(results)} slides: {[r['slide'] for r in results]}") return results
pythondef extract_timeline(pptx_path, date_pattern=r'\d{4}[年/\-]\d{1,2}'): """Extract date-tagged events from slide text.""" import re slides_data = extract_all_slides_text(pptx_path) events = [] for s in slides_data: for line in s['text'].split('\n'): if re.search(date_pattern, line): events.append({'slide': s['slide'], 'event': line.strip()}) return events
pythondef compute_ratio_from_pptx_table(pptx_path, numerator_col, denominator_col): """Example: compute ratio = col_A / col_B for all rows.""" tables = extract_pptx_tables(pptx_path) for item in tables: df = item['df'] # Try to find columns (flexible matching) num_col = next((c for c in df.columns if numerator_col in c), None) den_col = next((c for c in df.columns if denominator_col in c), None) if num_col and den_col: df[num_col] = pd.to_numeric(df[num_col].str.replace('人', '').str.strip(), errors='coerce') df[den_col] = pd.to_numeric(df[den_col].str.replace('人', '').str.strip(), errors='coerce') df['ratio'] = (df[num_col] / df[den_col] * 100).round(0).astype(str) + '%' print(df[['slide' if 'slide' in df.columns else df.columns[0], num_col, den_col, 'ratio']].to_string())
pythonpptx_path = "/mnt/data/report.pptx" # 1. Extract text from all slides slides_data = extract_all_slides_text(pptx_path) # 2. Caption image-only slides slides_data = caption_image_slides(pptx_path, slides_data) # 3. Combine all text for analysis all_text = '\n\n'.join( f"[Slide {s['slide']}]\n{s['text']}" for s in slides_data if s.get('text') ) # 4. Search or analyze results = find_in_pptx(pptx_path, '录用占比', slides_data) # 5. Extract tables if needed tables = extract_pptx_tables(pptx_path)
| Pitfall | Fix | |---------|-----| | Skip slides with no text → miss chart data | Flag needs_caption, render & caption (Method 4) | | shape.chart.plots[0].series fails → no data | Catch exception, fall back to captioning the slide | | Table columns misread (企业名 vs 岗位名) | Print headers + first 3 rows before computing; verify column meaning | | Only read first N slides | Always for slide in prs.slides — no index limit | | .ppt format → python-pptx can't open | Convert to .pptx via libreoffice first | | PPT has overlapping text boxes → garbled order | Sort shapes by top-left position: sorted(slide.shapes, key=lambda s: (s.top, s.left)) |
| Case | Status | Duration (ms) | Turns | Tokens | Tool calls | ||||||||
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| Without | With | Δ | Without | With | Δ | Without | With | Δ | Without | With | Δ | ||
case-01 | fail→fail | 6,061 | 9,037 | +49% | 1 | 1 | 0% | 344 | 3,785 | +1000% | 0 | 0 | — |
case-02 | fail→fail | 8,856 | 7,589 | -14% | 1 | 1 | 0% | 733 | 3,602 | +391% | 0 | 0 | — |
case-03 | fail→pass | 9,565 | 40,780 | +326% | 1 | 1 | 0% | 548 | 5,194 | +848% | 0 | 0 | — |
case-04 | pass→pass | 28,428 | 26,591 | -6% | 1 | 1 | 0% | 6,181 | 9,321 | +51% | 0 | 0 | — |
case-05 | pass→pass | 15,022 | 10,995 | -27% | 1 | 1 | 0% | 2,450 | 5,098 | +108% | 0 | 0 | — |
case-06 | fail→pass | 2,148 | 4,265 | +99% | 1 | 1 | 0% | 310 | 3,860 | +1145% | 0 | 0 | — |
case-07 | pass→pass | 7,514 | 4,980 | -34% | 1 | 1 | 0% | 1,538 | 4,163 | +171% | 0 | 0 | — |
case-08 | fail→pass | 13,546 | 14,187 | +5% | 1 | 1 | 0% | 2,569 | 5,304 | +106% | 0 | 0 | — |
case-09 | pass→pass | 21,425 | 11,103 | -48% | 1 | 1 | 0% | 3,294 | 5,537 | +68% | 0 | 0 | — |
case-10 | fail→pass | 11,399 | 9,017 | -21% | 1 | 1 | 0% | 2,126 | 4,983 | +134% | 0 | 0 | — |
case-11 | fail→pass | 11,053 | 10,034 | -9% | 1 | 1 | 0% | 2,093 | 4,673 | +123% | 0 | 0 | — |
case-16 | pass→pass | 13,905 | 7,397 | -47% | 1 | 1 | 0% | 2,127 | 4,264 | +100% | 0 | 0 | — |
case-12 | pass→pass | 8,838 | 22,766 | +158% | 1 | 1 | 0% | 1,408 | 7,194 | +411% | 0 | 0 | — |
case-13 | pass→pass | 11,839 | 8,599 | -27% | 1 | 1 | 0% | 2,619 | 4,869 | +86% | 0 | 0 | — |
case-14 | pass→pass | 12,180 | 8,314 | -32% | 1 | 1 | 0% | 2,237 | 4,684 | +109% | 0 | 0 | — |
case-15 | fail→pass | 18,183 | 9,944 | -45% | 1 | 1 | 0% | 2,702 | 5,064 | +87% | 0 | 0 | — |
case-17 | pass→fail | 6,154 | 4,844 | -21% | 1 | 1 | 0% | 1,297 | 4,082 | +215% | 0 | 0 | — |
case-18 | fail→pass | 25,230 | 6,383 | -75% | 1 | 1 | 0% | 2,078 | 4,259 | +105% | 0 | 0 | — |
case-19 | fail→pass | 12,570 | 6,383 | -49% | 1 | 1 | 0% | 2,362 | 4,342 | +84% | 0 | 0 | — |
case-20 | fail→fail | 10,472 | 5,932 | -43% | 1 | 1 | 0% | 1,487 | 4,359 | +193% | 0 | 0 | — |
case-21 | pass→pass | 11,206 | 11,715 | +5% | 1 | 1 | 0% | 2,247 | 4,986 | +122% | 0 | 0 | — |
case-22 | fail→pass | 19,027 | 16,393 | -14% | 1 | 1 | 0% | 3,426 | 6,688 | +95% | 0 | 0 | — |
DecimalAI ran this skill against gemini-3.6-flash twice over the same eval suite — once with the skill loaded and once without — and compared the two runs case by case. 22 cases were attempted, and 19 counted toward the lift figure. The other 3 produced results that are not comparable between the two arms, so they are excluded from the headline rather than averaged into it. The headline lift of +36 percentage points is the difference between those two pass rates over the 19 comparable cases. 1 case got worse with the skill loaded, and it is included in that figure.
Without the skill loaded, the model failed this case. With it loaded, the same prompt on the same model passed. This is one improved case from the latest verified run; every case, including any that regressed, is in the table above.
Other measured skills in the registry, with their headline benchmark lift.