一、简介
一般Word文件后缀有doc、docx两种。docx是office word 2007以及以后版本文档的扩展名;doc是office word 2003文档保存的扩展名。对于这两种格式的word转换成html需要使用不同的方法。
对于docx格式的文档使用xdocreport进行转换。依赖如下:
<dependency><groupId>fr.opensagres.xdocreport</groupId><artifactId>fr.opensagres.xdocreport.document</artifactId><version>1.0.5</version></dependency><dependency><groupId>fr.opensagres.xdocreport</groupId><artifactId>org.apache.poi.xwpf.converter.xhtml</artifactId><version>1.0.5</version></dependency>
对于docx格式的文档使用poi进行转换。依赖如下:
<dependency><groupId>org.apache.poi</groupId><artifactId>poi</artifactId><version>3.12</version></dependency><dependency><groupId>org.apache.poi</groupId><artifactId>poi-scratchpad</artifactId><version>3.12</version></dependency>
二:示例
代码示例如下:
1packagecom.test.word;23importjava.io.File;4importjava.io.FileInputStream;5importjava.io.FileNotFoundException;6importjava.io.FileOutputStream;7importjava.io.IOException;8importjava.io.InputStream;9importjava.io.OutputStream;1011importjavax.xml.parsers.DocumentBuilderFactory;12importjavax.xml.parsers.ParserConfigurationException;13importjavax.xml.transform.OutputKeys;14importjavax.xml.transform.Transformer;15importjavax.xml.transform.TransformerException;16importjavax.xml.transform.TransformerFactory;17importjavax.xml.transform.dom.DOMSource;18importjavax.xml.transform.stream.StreamResult;1920importorg.apache.poi.hwpf.HWPFDocument;21importorg.apache.poi.hwpf.converter.PicturesManager;22importorg.apache.poi.hwpf.converter.WordToHtmlConverter;23importorg.apache.poi.hwpf.usermodel.PictureType;24importorg.apache.poi.xwpf.converter.core.FileImageExtractor;25importorg.apache.poi.xwpf.converter.core.FileURIResolver;26importorg.apache.poi.xwpf.converter.xhtml.XHTMLConverter;27importorg.apache.poi.xwpf.converter.xhtml.XHTMLOptions;28importorg.apache.poi.xwpf.usermodel.XWPFDocument;29importorg.junit.Test;30importorg.w3c.dom.Document;3132/**33* word 转换成html34*/35publicclassWordToHtml {3637/**38* 2007版本word转换成html39*@throwsIOException40*/41@Test42publicvoidWord2007ToHtml()throwsIOException {43String filepath = "C:/test/";44String fileName = "滕王阁序2007.docx";45String htmlName = "滕王阁序2007.html";46finalString file = filepath +fileName;47File f =newFile(file);48if(!f.exists()) {49System.out.PRintln("Sorry File does not Exists!");50}else{51if(f.getName().endsWith(".docx") || f.getName().endsWith(".DOCX")) {5253//1) 加载word文档生成 XWPFDocument对象54InputStream in =newFileInputStream(f);55XWPFDocument document =newXWPFDocument(in);5657//2) 解析 XHTML配置 (这里设置IURIResolver来设置图片存放的目录)58File imageFolderFile =newFile(filepath);59XHTMLOptions options = XHTMLOptions.create().URIResolver(newFileURIResolver(imageFolderFile));60options.setExtractor(newFileImageExtractor(imageFolderFile));61options.setIgnoreStylesIfUnused(false);62options.setFragment(true);6364//3) 将 XWPFDocument转换成XHTML65OutputStream out =newFileOutputStream(newFile(filepath +htmlName));66XHTMLConverter.getInstance().convert(document, out, options);6768//也可以使用字符数组流获取解析的内容69//ByteArrayOutputStream baos = new ByteArrayOutputStream();70//XHTMLConverter.getInstance().convert(document, baos, options);71//String content = baos.toString();72//System.out.println(content);73//baos.close();74}else{75System.out.println("Enter only MS Office 2007+ files");76}77}78}7980/**81* /**82* 2003版本word转换成html83*@throwsIOException84*@throwsTransformerException85*@throwsParserConfigurationException86*/87@Test88publicvoidWord2003ToHtml()throwsIOException, TransformerException, ParserConfigurationException {89String filepath = "C:/test/";90finalString imagepath = "C:/test/image/";91String fileName = "滕王阁序2003.doc";92String htmlName = "滕王阁序2003.html";93finalString file = filepath +fileName;94InputStream input =newFileInputStream(newFile(file));95HWPFDocument wordDocument =newHWPFDocument(input);96WordToHtmlConverter wordToHtmlConverter =newWordToHtmlConverter(DocumentBuilderFactory.newInstance().newDocumentBuilder().newDocument());97//设置图片存放的位置98wordToHtmlConverter.setPicturesManager(newPicturesManager() {99publicString savePicture(byte[] content, PictureType pictureType, String suggestedName,floatwidthInches,floatheightInches) {100File imgPath =newFile(imagepath);101if(!imgPath.exists()){//图片目录不存在则创建102imgPath.mkdirs();103}104File file =newFile(imagepath +suggestedName);105try{106OutputStream os =newFileOutputStream(file);107os.write(content);108os.close();109}catch(FileNotFoundException e) {110e.printStackTrace();111}catch(IOException e) {112e.printStackTrace();113}114returnimagepath +suggestedName;115}116});117118//解析word文档119wordToHtmlConverter.processDocument(wordDocument);120Document htmlDocument =wordToHtmlConverter.getDocument();121122File htmlFile =newFile(filepath +htmlName);123OutputStream outStream =newFileOutputStream(htmlFile);124125//也可以使用字符数组流获取解析的内容126//ByteArrayOutputStream baos = new ByteArrayOutputStream();127//OutputStream outStream = new BufferedOutputStream(baos);128129DOMSource domSource =newDOMSource(htmlDocument);130StreamResult streamResult =newStreamResult(outStream);131132TransformerFactory factory =TransformerFactory.newInstance();133Transformer serializer =factory.newTransformer();134serializer.setOutputProperty(OutputKeys.ENCODING, "utf-8");135serializer.setOutputProperty(OutputKeys.INDENT, "yes");136serializer.setOutputProperty(OutputKeys.METHOD, "html");137138serializer.transform(domSource, streamResult);139140//也可以使用字符数组流获取解析的内容141//String content = baos.toString();142//System.out.println(content);143//baos.close();144outStream.close();145}146}
运行生存文件结果如下: