Last active
August 29, 2015 14:17
-
-
Save kabronkline/e9f2c0fcad02c69c3212 to your computer and use it in GitHub Desktop.
Split a Large Text File Into Smaller Text Files By Chunk Size Extremely Fast
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| /* | |
| * Copyright (c) 1996, 2013, Oracle and/or its affiliates. All rights reserved. | |
| * ORACLE PROPRIETARY/CONFIDENTIAL. Use is subject to license terms. | |
| * | |
| * | |
| * | |
| * | |
| * | |
| * | |
| * | |
| * | |
| * | |
| * | |
| * | |
| * | |
| * | |
| * | |
| * | |
| * | |
| * | |
| * | |
| * | |
| * | |
| */ | |
| package com.kabronkline.utility.file; | |
| import java.io.BufferedReader; | |
| import java.io.IOException; | |
| import java.io.Reader; | |
| import java.io.UncheckedIOException; | |
| import java.util.Iterator; | |
| import java.util.NoSuchElementException; | |
| import java.util.Spliterator; | |
| import java.util.Spliterators; | |
| import java.util.stream.Stream; | |
| import java.util.stream.StreamSupport; | |
| public class LineTerminatorTrackingBufferedReader extends BufferedReader { | |
| private Reader in; | |
| private char cb[]; | |
| private int nChars, nextChar; | |
| private char lastLineTerminatorChars[]; | |
| private static final int INVALIDATED = -2; | |
| private static final int UNMARKED = -1; | |
| private int markedChar = UNMARKED; | |
| private int readAheadLimit = 0; /* Valid only when markedChar > 0 */ | |
| /** If the next character is a line feed, skip it */ | |
| private boolean skipLF = false; | |
| /** The skipLF flag when the mark was set */ | |
| private boolean markedSkipLF = false; | |
| private static int defaultCharBufferSize = 8192; | |
| private static int defaultExpectedLineLength = 80; | |
| /** | |
| * Creates a buffering character-input stream that uses an input buffer of | |
| * the specified size. | |
| * | |
| * @param in A Reader | |
| * @param sz Input-buffer size | |
| * | |
| * @exception IllegalArgumentException If {@code sz <= 0} | |
| */ | |
| public LineTerminatorTrackingBufferedReader(Reader in, int sz) { | |
| super(in); | |
| if (sz <= 0) | |
| throw new IllegalArgumentException("Buffer size <= 0"); | |
| this.in = in; | |
| cb = new char[sz]; | |
| nextChar = nChars = 0; | |
| lastLineTerminatorChars = new char[2]; | |
| } | |
| /** | |
| * Creates a buffering character-input stream that uses a default-sized | |
| * input buffer. | |
| * | |
| * @param in A Reader | |
| */ | |
| public LineTerminatorTrackingBufferedReader(Reader in) { | |
| this(in, defaultCharBufferSize); | |
| } | |
| /** Checks to make sure that the stream has not been closed */ | |
| private void ensureOpen() throws IOException { | |
| if (in == null) | |
| throw new IOException("Stream closed"); | |
| } | |
| /** | |
| * Fills the input buffer, taking the mark into account if it is valid. | |
| */ | |
| private void fill() throws IOException { | |
| int dst; | |
| if (markedChar <= UNMARKED) { | |
| /* No mark */ | |
| dst = 0; | |
| } else { | |
| /* Marked */ | |
| int delta = nextChar - markedChar; | |
| if (delta >= readAheadLimit) { | |
| /* Gone past read-ahead limit: Invalidate mark */ | |
| markedChar = INVALIDATED; | |
| readAheadLimit = 0; | |
| dst = 0; | |
| } else { | |
| if (readAheadLimit <= cb.length) { | |
| /* Shuffle in the current buffer */ | |
| System.arraycopy(cb, markedChar, cb, 0, delta); | |
| markedChar = 0; | |
| dst = delta; | |
| } else { | |
| /* Reallocate buffer to accommodate read-ahead limit */ | |
| char ncb[] = new char[readAheadLimit]; | |
| System.arraycopy(cb, markedChar, ncb, 0, delta); | |
| cb = ncb; | |
| markedChar = 0; | |
| dst = delta; | |
| } | |
| nextChar = nChars = delta; | |
| } | |
| } | |
| int n; | |
| do { | |
| n = in.read(cb, dst, cb.length - dst); | |
| } while (n == 0); | |
| if (n > 0) { | |
| nChars = dst + n; | |
| nextChar = dst; | |
| } | |
| } | |
| /** | |
| * Reads a single character. | |
| * | |
| * @return The character read, as an integer in the range | |
| * 0 to 65535 (<tt>0x00-0xffff</tt>), or -1 if the | |
| * end of the stream has been reached | |
| * @exception IOException If an I/O error occurs | |
| */ | |
| public int read() throws IOException { | |
| synchronized (lock) { | |
| ensureOpen(); | |
| for (;;) { | |
| if (nextChar >= nChars) { | |
| fill(); | |
| if (nextChar >= nChars) | |
| return -1; | |
| } | |
| if (skipLF) { | |
| skipLF = false; | |
| if (cb[nextChar] == '\n') { | |
| nextChar++; | |
| continue; | |
| } | |
| } | |
| return cb[nextChar++]; | |
| } | |
| } | |
| } | |
| /** | |
| * Reads characters into a portion of an array, reading from the underlying | |
| * stream if necessary. | |
| */ | |
| private int read1(char[] cbuf, int off, int len) throws IOException { | |
| if (nextChar >= nChars) { | |
| /* If the requested length is at least as large as the buffer, and | |
| if there is no mark/reset activity, and if line feeds are not | |
| being skipped, do not bother to copy the characters into the | |
| local buffer. In this way buffered streams will cascade | |
| harmlessly. */ | |
| if (len >= cb.length && markedChar <= UNMARKED && !skipLF) { | |
| return in.read(cbuf, off, len); | |
| } | |
| fill(); | |
| } | |
| if (nextChar >= nChars) return -1; | |
| if (skipLF) { | |
| skipLF = false; | |
| if (cb[nextChar] == '\n') { | |
| nextChar++; | |
| if (nextChar >= nChars) | |
| fill(); | |
| if (nextChar >= nChars) | |
| return -1; | |
| } | |
| } | |
| int n = Math.min(len, nChars - nextChar); | |
| System.arraycopy(cb, nextChar, cbuf, off, n); | |
| nextChar += n; | |
| return n; | |
| } | |
| /** | |
| * Reads characters into a portion of an array. | |
| * | |
| * <p> This method implements the general contract of the corresponding | |
| * <code>{@link Reader#read(char[], int, int) read}</code> method of the | |
| * <code>{@link Reader}</code> class. As an additional convenience, it | |
| * attempts to read as many characters as possible by repeatedly invoking | |
| * the <code>read</code> method of the underlying stream. This iterated | |
| * <code>read</code> continues until one of the following conditions becomes | |
| * true: <ul> | |
| * | |
| * <li> The specified number of characters have been read, | |
| * | |
| * <li> The <code>read</code> method of the underlying stream returns | |
| * <code>-1</code>, indicating end-of-file, or | |
| * | |
| * <li> The <code>ready</code> method of the underlying stream | |
| * returns <code>false</code>, indicating that further input requests | |
| * would block. | |
| * | |
| * </ul> If the first <code>read</code> on the underlying stream returns | |
| * <code>-1</code> to indicate end-of-file then this method returns | |
| * <code>-1</code>. Otherwise this method returns the number of characters | |
| * actually read. | |
| * | |
| * <p> Subclasses of this class are encouraged, but not required, to | |
| * attempt to read as many characters as possible in the same fashion. | |
| * | |
| * <p> Ordinarily this method takes characters from this stream's character | |
| * buffer, filling it from the underlying stream as necessary. If, | |
| * however, the buffer is empty, the mark is not valid, and the requested | |
| * length is at least as large as the buffer, then this method will read | |
| * characters directly from the underlying stream into the given array. | |
| * Thus redundant <code>BufferedReader</code>s will not copy data | |
| * unnecessarily. | |
| * | |
| * @param cbuf Destination buffer | |
| * @param off Offset at which to start storing characters | |
| * @param len Maximum number of characters to read | |
| * | |
| * @return The number of characters read, or -1 if the end of the | |
| * stream has been reached | |
| * | |
| * @exception IOException If an I/O error occurs | |
| */ | |
| public int read(char cbuf[], int off, int len) throws IOException { | |
| synchronized (lock) { | |
| ensureOpen(); | |
| if ((off < 0) || (off > cbuf.length) || (len < 0) || | |
| ((off + len) > cbuf.length) || ((off + len) < 0)) { | |
| throw new IndexOutOfBoundsException(); | |
| } else if (len == 0) { | |
| return 0; | |
| } | |
| int n = read1(cbuf, off, len); | |
| if (n <= 0) return n; | |
| while ((n < len) && in.ready()) { | |
| int n1 = read1(cbuf, off + n, len - n); | |
| if (n1 <= 0) break; | |
| n += n1; | |
| } | |
| return n; | |
| } | |
| } | |
| /** | |
| * Reads a line of text. A line is considered to be terminated by any one | |
| * of a line feed ('\n'), a carriage return ('\r'), or a carriage return | |
| * followed immediately by a linefeed. | |
| * | |
| * @param ignoreLF If true, the next '\n' will be skipped | |
| * | |
| * @return A String containing the contents of the line, not including | |
| * any line-termination characters, or null if the end of the | |
| * stream has been reached | |
| * | |
| * @see java.io.LineNumberReader#readLine() | |
| * | |
| * @exception IOException If an I/O error occurs | |
| */ | |
| String readLine(boolean ignoreLF) throws IOException { | |
| StringBuffer s = null; | |
| int startChar; | |
| synchronized (lock) { | |
| ensureOpen(); | |
| boolean omitLF = ignoreLF || skipLF; | |
| bufferLoop: | |
| for (;;) { | |
| if (nextChar >= nChars) | |
| fill(); | |
| if (nextChar >= nChars) { /* EOF */ | |
| if (s != null && s.length() > 0) { | |
| lastLineTerminatorChars = new char[0]; | |
| return s.toString(); | |
| } | |
| else | |
| return null; | |
| } | |
| boolean eol = false; | |
| char c = 0; | |
| int i; | |
| /* Skip a leftover '\n', if necessary */ | |
| if (omitLF && (cb[nextChar] == '\n')) { | |
| nextChar++; | |
| } | |
| skipLF = false; | |
| omitLF = false; | |
| charLoop: | |
| for (i = nextChar; i < nChars; i++) { | |
| c = cb[i]; | |
| if ((c == '\n') || (c == '\r')) { | |
| eol = true; | |
| lastLineTerminatorChars[0] = c; | |
| break charLoop; | |
| } | |
| } | |
| startChar = nextChar; | |
| nextChar = i; | |
| if (eol) { | |
| String str; | |
| if (s == null) { | |
| str = new String(cb, startChar, i - startChar); | |
| } else { | |
| s.append(cb, startChar, i - startChar); | |
| str = s.toString(); | |
| } | |
| nextChar++; | |
| if (c == '\r') { | |
| if (nextChar < nChars) lastLineTerminatorChars[1] = cb[nextChar]; | |
| skipLF = true; | |
| } | |
| return str; | |
| } | |
| if (s == null) | |
| s = new StringBuffer(defaultExpectedLineLength); | |
| s.append(cb, startChar, i - startChar); | |
| } | |
| } | |
| } | |
| /** | |
| * Reads a line of text. A line is considered to be terminated by any one | |
| * of a line feed ('\n'), a carriage return ('\r'), or a carriage return | |
| * followed immediately by a linefeed. | |
| * | |
| * @return A String containing the contents of the line, not including | |
| * any line-termination characters, or null if the end of the | |
| * stream has been reached | |
| * | |
| * @exception IOException If an I/O error occurs | |
| * | |
| * @see java.nio.file.Files#readAllLines | |
| */ | |
| public String readLine() throws IOException { | |
| return readLine(false); | |
| } | |
| /** | |
| * Skips characters. | |
| * | |
| * @param n The number of characters to skip | |
| * | |
| * @return The number of characters actually skipped | |
| * | |
| * @exception IllegalArgumentException If <code>n</code> is negative. | |
| * @exception IOException If an I/O error occurs | |
| */ | |
| public long skip(long n) throws IOException { | |
| if (n < 0L) { | |
| throw new IllegalArgumentException("skip value is negative"); | |
| } | |
| synchronized (lock) { | |
| ensureOpen(); | |
| long r = n; | |
| while (r > 0) { | |
| if (nextChar >= nChars) | |
| fill(); | |
| if (nextChar >= nChars) /* EOF */ | |
| break; | |
| if (skipLF) { | |
| skipLF = false; | |
| if (cb[nextChar] == '\n') { | |
| nextChar++; | |
| } | |
| } | |
| long d = nChars - nextChar; | |
| if (r <= d) { | |
| nextChar += r; | |
| r = 0; | |
| break; | |
| } | |
| else { | |
| r -= d; | |
| nextChar = nChars; | |
| } | |
| } | |
| return n - r; | |
| } | |
| } | |
| /** | |
| * Tells whether this stream is ready to be read. A buffered character | |
| * stream is ready if the buffer is not empty, or if the underlying | |
| * character stream is ready. | |
| * | |
| * @exception IOException If an I/O error occurs | |
| */ | |
| public boolean ready() throws IOException { | |
| synchronized (lock) { | |
| ensureOpen(); | |
| /* | |
| * If newline needs to be skipped and the next char to be read | |
| * is a newline character, then just skip it right away. | |
| */ | |
| if (skipLF) { | |
| /* Note that in.ready() will return true if and only if the next | |
| * read on the stream will not block. | |
| */ | |
| if (nextChar >= nChars && in.ready()) { | |
| fill(); | |
| } | |
| if (nextChar < nChars) { | |
| if (cb[nextChar] == '\n') | |
| nextChar++; | |
| skipLF = false; | |
| } | |
| } | |
| return (nextChar < nChars) || in.ready(); | |
| } | |
| } | |
| /** | |
| * Tells whether this stream supports the mark() operation, which it does. | |
| */ | |
| public boolean markSupported() { | |
| return true; | |
| } | |
| /** | |
| * Marks the present position in the stream. Subsequent calls to reset() | |
| * will attempt to reposition the stream to this point. | |
| * | |
| * @param readAheadLimit Limit on the number of characters that may be | |
| * read while still preserving the mark. An attempt | |
| * to reset the stream after reading characters | |
| * up to this limit or beyond may fail. | |
| * A limit value larger than the size of the input | |
| * buffer will cause a new buffer to be allocated | |
| * whose size is no smaller than limit. | |
| * Therefore large values should be used with care. | |
| * | |
| * @exception IllegalArgumentException If {@code readAheadLimit < 0} | |
| * @exception IOException If an I/O error occurs | |
| */ | |
| public void mark(int readAheadLimit) throws IOException { | |
| if (readAheadLimit < 0) { | |
| throw new IllegalArgumentException("Read-ahead limit < 0"); | |
| } | |
| synchronized (lock) { | |
| ensureOpen(); | |
| this.readAheadLimit = readAheadLimit; | |
| markedChar = nextChar; | |
| markedSkipLF = skipLF; | |
| } | |
| } | |
| /** | |
| * Resets the stream to the most recent mark. | |
| * | |
| * @exception IOException If the stream has never been marked, | |
| * or if the mark has been invalidated | |
| */ | |
| public void reset() throws IOException { | |
| synchronized (lock) { | |
| ensureOpen(); | |
| if (markedChar < 0) | |
| throw new IOException((markedChar == INVALIDATED) | |
| ? "Mark invalid" | |
| : "Stream not marked"); | |
| nextChar = markedChar; | |
| skipLF = markedSkipLF; | |
| } | |
| } | |
| public void close() throws IOException { | |
| synchronized (lock) { | |
| if (in == null) | |
| return; | |
| try { | |
| in.close(); | |
| } finally { | |
| in = null; | |
| cb = null; | |
| } | |
| } | |
| } | |
| /** | |
| * Returns a {@code Stream}, the elements of which are lines read from | |
| * this {@code BufferedReader}. The {@link Stream} is lazily populated, | |
| * i.e., read only occurs during the | |
| * <a href="../util/stream/package-summary.html#StreamOps">terminal | |
| * stream operation</a>. | |
| * | |
| * <p> The reader must not be operated on during the execution of the | |
| * terminal stream operation. Otherwise, the result of the terminal stream | |
| * operation is undefined. | |
| * | |
| * <p> After execution of the terminal stream operation there are no | |
| * guarantees that the reader will be at a specific position from which to | |
| * read the next character or line. | |
| * | |
| * <p> If an {@link IOException} is thrown when accessing the underlying | |
| * {@code BufferedReader}, it is wrapped in an {@link | |
| * UncheckedIOException} which will be thrown from the {@code Stream} | |
| * method that caused the read to take place. This method will return a | |
| * Stream if invoked on a BufferedReader that is closed. Any operation on | |
| * that stream that requires reading from the BufferedReader after it is | |
| * closed, will cause an UncheckedIOException to be thrown. | |
| * | |
| * @return a {@code Stream<String>} providing the lines of text | |
| * described by this {@code BufferedReader} | |
| * | |
| * @since 1.8 | |
| */ | |
| public Stream<String> lines() { | |
| Iterator<String> iter = new Iterator<String>() { | |
| String nextLine = null; | |
| @Override | |
| public boolean hasNext() { | |
| if (nextLine != null) { | |
| return true; | |
| } else { | |
| try { | |
| nextLine = readLine(); | |
| return (nextLine != null); | |
| } catch (IOException e) { | |
| throw new UncheckedIOException(e); | |
| } | |
| } | |
| } | |
| @Override | |
| public String next() { | |
| if (nextLine != null || hasNext()) { | |
| String line = nextLine; | |
| nextLine = null; | |
| return line; | |
| } else { | |
| throw new NoSuchElementException(); | |
| } | |
| } | |
| }; | |
| return StreamSupport.stream(Spliterators.spliteratorUnknownSize( | |
| iter, Spliterator.ORDERED | Spliterator.NONNULL), false); | |
| } | |
| /** | |
| * <p>Returns a char array containing at most two characters | |
| * which identifies the line terminator of the last line returned | |
| * by a call to the readLine() method. | |
| * | |
| * @return char array containing one of four possibilities detailed below. | |
| * | |
| * <p>1. For lines ending with a Carriage Return character ('\r') : | |
| * | |
| * <p>char[0] = '\r' | |
| * | |
| * <p>2. For lines ending with a Line Feed character ('\n') : | |
| * | |
| * <p>char[0] = '\n' | |
| * | |
| * <p>3. For lines ending with Carriage Return character | |
| * immediately followed by a Line Feed character ('\r\n') : | |
| * | |
| * <p>char[0] = '\r' | |
| * <p>char[1] = '\n' | |
| * | |
| * <p>4. For line without any line terminator (i.e. last line in the file) | |
| * an empty char array will be returned. | |
| */ | |
| public char[] getLastLineTerminatorChars() { | |
| return lastLineTerminatorChars; | |
| } | |
| } |
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| <project xmlns="http://maven.apache.org/POM/4.0.0" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://maven.apache.org/POM/4.0.0 http://maven.apache.org/xsd/maven-4.0.0.xsd"> | |
| <modelVersion>4.0.0</modelVersion> | |
| <groupId>com.kabronkline</groupId> | |
| <artifactId>utility</artifactId> | |
| <version>0.0.1-SNAPSHOT</version> | |
| <build> | |
| <sourceDirectory>src</sourceDirectory> | |
| <testSourceDirectory>test</testSourceDirectory> | |
| <plugins> | |
| <plugin> | |
| <artifactId>maven-compiler-plugin</artifactId> | |
| <version>3.1</version> | |
| <configuration> | |
| <source>1.7</source> | |
| <target>1.7</target> | |
| </configuration> | |
| </plugin> | |
| </plugins> | |
| </build> | |
| <dependencies> | |
| <dependency> | |
| <groupId>commons-io</groupId> | |
| <artifactId>commons-io</artifactId> | |
| <version>2.4</version> | |
| </dependency> | |
| </dependencies> | |
| </project> |
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| package com.kabronkline.utility.file; | |
| import java.io.File; | |
| import java.io.FileInputStream; | |
| import java.io.InputStreamReader; | |
| import java.io.RandomAccessFile; | |
| import java.nio.ByteBuffer; | |
| import java.nio.MappedByteBuffer; | |
| import java.nio.channels.FileChannel; | |
| import java.nio.channels.FileChannel.MapMode; | |
| import java.nio.charset.Charset; | |
| import java.nio.charset.CharsetDecoder; | |
| import java.nio.charset.CharsetEncoder; | |
| import java.util.HashMap; | |
| import java.util.HashSet; | |
| import java.util.Map; | |
| import java.util.Set; | |
| import org.apache.commons.io.FilenameUtils; | |
| import org.apache.commons.io.IOUtils; | |
| import org.apache.commons.io.LineIterator; | |
| /** | |
| * Supports splitting text files into several smaller text files | |
| * in an extremely efficient (i.e. FAST) manner. | |
| * | |
| * @author Kabron Kline | |
| * | |
| */ | |
| public class TextFileSplitter { | |
| /** | |
| * Given a source file, the charset encoding of the containing text, and | |
| * the maximum split size allowed this method will split the source file | |
| * into several individual files in byte sizes less than or equal to the | |
| * splitMaxByteSize parameter while preserving integrity of individual lines. | |
| * | |
| * Note that since line integrity is guaranteed the individual split files | |
| * can be of different sizes since file splits always begin with a full line | |
| * and end with a full line (i.e. no lines will be split across separate files). | |
| * | |
| * @param sourceTextFile The source file which contains textual data | |
| * @param encoding The charset encoding of the textual data in the source file | |
| * @param splitMaxByteSize The maximum size in bytes of the individual split files | |
| * @return The set of split files created from the source file | |
| * @throws Exception | |
| */ | |
| public Set<File> splitTextFile(File sourceTextFile, Charset encoding, | |
| long splitMaxByteSize) throws Exception | |
| { | |
| long sourceTotalByteSize = sourceTextFile.length(); | |
| Set<File> splitFiles = new HashSet<File>(); | |
| if (sourceTotalByteSize <= splitMaxByteSize) { | |
| splitFiles.add(sourceTextFile); | |
| return splitFiles; | |
| } | |
| // Calculate total number of split files | |
| int numSplitFiles = ((int)(sourceTotalByteSize / splitMaxByteSize)) + | |
| (splitMaxByteSize % sourceTotalByteSize > 0 ? 1 : 0); | |
| /** | |
| * Maps the split file number to a byte index in the source file that | |
| * represents the last byte stored in the split file. | |
| */ | |
| Map<Integer, Long> splitNumLastLineTotalBytes = new HashMap<Integer, Long>(); | |
| Long totalByteSize = 0L; | |
| LineIterator it = null; | |
| FileInputStream sourceInputStream = new FileInputStream(sourceTextFile); | |
| int lastLineBreakByteSize = 0; | |
| try { | |
| LineTerminatorTrackingBufferedReader advancedBufferReader = | |
| new LineTerminatorTrackingBufferedReader( | |
| new InputStreamReader(sourceInputStream, encoding)); | |
| it = IOUtils.lineIterator(advancedBufferReader); | |
| while (it.hasNext()) { | |
| char[] lastLineTerminators = advancedBufferReader.getLastLineTerminatorChars(); | |
| lastLineBreakByteSize = new String(lastLineTerminators).getBytes(encoding).length; | |
| totalByteSize += it.nextLine().getBytes(encoding).length + lastLineBreakByteSize; | |
| int splitFileNum = (int)(totalByteSize / splitMaxByteSize) + 1; | |
| splitNumLastLineTotalBytes.put(splitFileNum, totalByteSize); | |
| } | |
| } finally { | |
| LineIterator.closeQuietly(it); | |
| } | |
| FileChannel sourceFileChannel = null; | |
| try (RandomAccessFile sourceTextFileRAM = new RandomAccessFile(sourceTextFile, "r")) { | |
| sourceFileChannel = sourceTextFileRAM.getChannel(); | |
| int position = 0; | |
| Long lastTotalBytesOfLastLineForSplitFile = 0L; | |
| for (int i = 1; i <= numSplitFiles; i++) { | |
| Long totalBytesOfLastLineForSplitFile = splitNumLastLineTotalBytes.get(i); | |
| Long byteSize = totalBytesOfLastLineForSplitFile - lastTotalBytesOfLastLineForSplitFile; | |
| MappedByteBuffer mappedByteBuffer = sourceFileChannel.map( | |
| MapMode.READ_ONLY, position, byteSize); | |
| position = totalBytesOfLastLineForSplitFile.intValue(); | |
| lastTotalBytesOfLastLineForSplitFile = totalBytesOfLastLineForSplitFile; | |
| File splitFile = getSplitFile(sourceTextFile, i); | |
| if (splitFile.exists()) splitFile.delete(); | |
| try (RandomAccessFile splitFileRAM = new RandomAccessFile(splitFile, "rw")) { | |
| FileChannel splitFileChannel = splitFileRAM.getChannel(); | |
| splitFileChannel.write(mappedByteBuffer); | |
| splitFiles.add(splitFile); | |
| } | |
| } | |
| } | |
| return splitFiles; | |
| } | |
| private File getSplitFile(File sourceTextFile, int splitNum) { | |
| String splitFileName = FilenameUtils.getBaseName(sourceTextFile.getName()) + "_" + splitNum + "." + | |
| FilenameUtils.getExtension(sourceTextFile.getName()); | |
| File splitFile = new File(sourceTextFile.getParent(), splitFileName); | |
| return splitFile; | |
| } | |
| } |
Author
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment
Machine Specs: i7-2600K @ 4.7 GHz, Samsung 840 EVO SSD (AHCI mode), 16 GB DDR3 1800
Splits a 450 MB file into 100 MB chunks in 2.3 seconds.