java流式处理zip+多线程

概述

流式处理一个zip,zip里有多个json文件。

流式处理可以避免解压一个大的zip。再加上多线程,处理的效率杠杠的。

代码

java 复制代码
package 多线程.demo05多jsonCountDownLatch;

import com.fasterxml.jackson.databind.ObjectMapper;
import lombok.SneakyThrows;
import lombok.extern.slf4j.Slf4j;
import org.springframework.util.StopWatch;

import java.io.ByteArrayOutputStream;
import java.io.FileNotFoundException;
import java.io.IOException;
import java.nio.file.Files;
import java.nio.file.Path;
import java.nio.file.Paths;
import java.util.ArrayList;
import java.util.List;
import java.util.concurrent.ExecutorService;
import java.util.concurrent.Executors;
import java.util.concurrent.Future;
import java.util.concurrent.TimeUnit;
import java.util.zip.ZipEntry;
import java.util.zip.ZipInputStream;

@Slf4j
public class ZipProcessor {

    private Path path;

    private static int numThreads = Runtime.getRuntime().availableProcessors();
    private static ExecutorService executorService = Executors.newFixedThreadPool(numThreads);
    private static ObjectMapper objectMapper = new ObjectMapper();

    @SneakyThrows
    public ZipProcessor(String filePath){
        path = Paths.get(filePath);
        if (!Files.exists(path)) {
            throw new FileNotFoundException("The specified ZIP file does not exist: " + filePath);
        }
    }

    public void streamProcess(){
        StopWatch stopWatch = new StopWatch();
        stopWatch.start();
        // 使用 try-with-resources 保证资源关闭
        try (ZipInputStream zis = new ZipInputStream(Files.newInputStream(path))) {
            ZipEntry entry;
            while ((entry = zis.getNextEntry()) != null) {
                // 将当前条目的数据读取到字节数组中
                byte[] byteArray = getByteArray(zis);
                process(byteArray, entry);
                // 关闭当前条目的输入流
                zis.closeEntry();
            }
            stopWatch.stop();
            log.info("zip处理耗时:{}秒", stopWatch.getTotalTimeSeconds());
        } catch (IOException e) {
            stopWatch.stop();
            log.error("zip处理异常,耗时:{}秒", stopWatch.getTotalTimeSeconds(), e);
        }

    }

    public void streamParallelProcess() {
        StopWatch stopWatch = new StopWatch();
        stopWatch.start();
        try (ZipInputStream zis = new ZipInputStream(Files.newInputStream(path))) {
            ZipEntry entry;
            List<Future<?>> futures = new ArrayList<>();
            while ((entry = zis.getNextEntry()) != null) {
                // 为了lambda表达式捕获局部变量
                final ZipEntry currentEntry = entry;
                // 将当前条目的数据读取到字节数组中
                byte[] byteArray = getByteArray(zis);
                Future<?> future = executorService.submit(() -> process(byteArray, currentEntry));
                futures.add(future);
            }

            // 等待所有任务完成
            for (Future<?> future : futures) {
                try {
                    future.get();
                } catch (Exception e) {
                    log.error("任务执行失败", e);
                }
            }
        } catch (IOException e) {
            log.error("读取ZIP文件异常", e);
        } finally {
            executorService.shutdown();
            try {
                if (!executorService.awaitTermination(60, TimeUnit.SECONDS)) {
                    executorService.shutdownNow();
                }
            } catch (InterruptedException ex) {
                executorService.shutdownNow();
            }
            stopWatch.stop();
            log.info("zip处理耗时:{}秒", stopWatch.getTotalTimeSeconds());
        }
    }

    private void process(byte[] entryData, ZipEntry entry) {
        try {
            // 在这里处理每个条目的数据
            ObjectMapper objectMapper = new ObjectMapper();
            OriginalObject originalObject = objectMapper.readValue(entryData, OriginalObject.class);
            log.info("完成处理:{},sourceFileId:{}", entry.getName(), originalObject.getSourceFileId());
        } catch (IOException e) {
            log.error("处理条目 {} 异常", entry.getName(), e);
        }
    }

    private byte[] getByteArray(ZipInputStream zis) throws IOException {
        ByteArrayOutputStream baos = new ByteArrayOutputStream();
        byte[] buffer = new byte[1024];
        int length;
        while ((length = zis.read(buffer)) > 0) {
            baos.write(buffer, 0, length);
        }
        return baos.toByteArray();
    }

}
相关推荐
荼蘼15 小时前
使用 Flask 实现本机 PyTorch 模型部署:从服务端搭建到客户端调用
人工智能·pytorch·python
LL_break16 小时前
Mysql数据库
java·数据库·mysql
白水先森16 小时前
Python 运算符与列表(list)
java·开发语言
野犬寒鸦16 小时前
从零起步学习Redis || 第十一章:主从切换时的哨兵机制如何实现及项目实战
java·服务器·数据库·redis·后端·缓存
(时光煮雨)16 小时前
【Python进阶】Python爬虫-Selenium
爬虫·python·selenium
爱读源码的大都督16 小时前
RAG效果不理想?试试用魔法打败魔法:让大模型深度参与优化的三阶段实战
java·人工智能·后端
小政同学16 小时前
【Python】小练习-考察变量作用域问题
开发语言·python
埃泽漫笔16 小时前
mq的常见问题
java·mq
Lynnxiaowen16 小时前
今天我们开始学习python3编程之python基础
linux·运维·python·学习
青青草原羊村懒大王16 小时前
1、pycharm相关知识
python