Showing posts with label map file. Show all posts
Showing posts with label map file. Show all posts

Monday, October 21, 2013

Hadoop : Merging Small tar files to the Sequence File

The Hadoop Distributed File System (HDFS) is a distributed file system. It is mainly designed for batch processing of large volume of data. The default block size of HDFS is 64MB. When data is represented in files significantly smaller than the default block size the performance degrades dramatically. Mainly there are two reasons for producing small files. One reason is some files are pieces of a larger logical file (e.g. - log files). Since HDFS has only recently supported appends, these unbounded files are saved by writing them in chunks into HDFS. Other reason is some files cannot be combined together into one larger file and are essentially small. e.g. - A large corpus of images where each image is a distinct file.

Solution to the small files by merging them into a Sequence File:

Sequence files is a Hadoop specific archive file format similar to tar and zip. The concept behind this is to merge the file set with using a key and a value pair and this created files known as ‘Hadoop Sequence Files’. In this method file name is used as the key and the file content is used as value.

In the proposed solution we will demonstrate how to write small files to the Sequence File and a Sequence file reader which will list the file name in Sequence File:

Setting up a local file system:
import org.apache.hadoop.conf.Configuration;
import org.apache.hadoop.fs.FileSystem;

public class LocalSetup {

    private FileSystem fileSystem;
    private Configuration config;

    
    public LocalSetup() throws Exception {
        config = new Configuration();

        
        config.set("fs.file.impl", "org.apache.hadoop.fs.LocalFileSystem");

        fileSystem = FileSystem.get(config);
        if (fileSystem.getConf() == null) {
                throw new Exception("LocalFileSystem configuration is null");
        }
    }

    
    public Configuration getConf() {
        return config;
    }

    
    public FileSystem getLocalFileSystem() {
        return fileSystem;
    }
}

In the next course of action we will setup a class which will read from the .tar.gz,.tgz,.tar.bz2 extension files and write it to the Sequence File with key as the name of file and value be the content of the file:
import org.apache.tools.bzip2.CBZip2InputStream;
import org.apache.tools.tar.TarEntry;
import org.apache.tools.tar.TarInputStream;
import org.apache.hadoop.conf.Configuration;
import org.apache.hadoop.fs.FileSystem;
import org.apache.hadoop.fs.LocalFileSystem;
import org.apache.hadoop.fs.Path;
import org.apache.hadoop.io.BytesWritable;
import org.apache.hadoop.io.SequenceFile;
import org.apache.hadoop.io.Text;

import java.io.File;
import java.io.FileInputStream;
import java.io.InputStream;
import java.io.IOException;
import java.util.zip.GZIPInputStream;



public class TarToSeqFile {

    private File inputFile;
    private File outputFile;
    private LocalSetup setup;

    
    public TarToSeqFile() throws Exception {
        setup = new LocalSetup();
    }

    
    public void setInput(File inputFile) {
        this.inputFile = inputFile;
    }

    public void setOutput(File outputFile) {
        this.outputFile = outputFile;
    }

    public void execute() throws Exception {
        TarInputStream input = null;
        SequenceFile.Writer output = null;
        try {
            input = openInputFile();
            output = openOutputFile();
            TarEntry entry;
            while ((entry = input.getNextEntry()) != null) {
                if (entry.isDirectory()) { continue; }
                String filename = entry.getName();
                byte[] data = TarToSeqFile.getBytes(input, entry.getSize());
                
                Text key = new Text(filename);
                BytesWritable value = new BytesWritable(data);
                output.append(key, value);
            }
        } finally {
            if (input != null) { input.close(); }
            if (output != null) { output.close(); }
        }
    }

    private TarInputStream openInputFile() throws Exception {
        InputStream fileStream = new FileInputStream(inputFile);
        String name = inputFile.getName();
        InputStream theStream = null;
        if (name.endsWith(".tar.gz") || name.endsWith(".tgz")) {
            theStream = new GZIPInputStream(fileStream);
        } else if (name.endsWith(".tar.bz2") || name.endsWith(".tbz2")) {
            fileStream.skip(2);
            theStream = new CBZip2InputStream(fileStream);
        } else {
            theStream = fileStream;
        }
        return new TarInputStream(theStream);
    }

    private SequenceFile.Writer openOutputFile() throws Exception {
        Path outputPath = new Path(outputFile.getAbsolutePath());
        return SequenceFile.createWriter(setup.getLocalFileSystem(), setup.getConf(),
                                         outputPath,
                                         Text.class, BytesWritable.class,
                                         SequenceFile.CompressionType.BLOCK);
    }

    
    private static byte[] getBytes(TarInputStream input, long size) throws Exception {
        if (size > Integer.MAX_VALUE) {
            throw new Exception("A file in the tar archive is too large.");
        }
        int length = (int)size;
        byte[] bytes = new byte[length];

        int offset = 0;
        int numRead = 0;

        while (offset < bytes.length &&
               (numRead = input.read(bytes, offset, bytes.length - offset)) >= 0) {
            offset += numRead;
        }

        if (offset < bytes.length) {
            throw new IOException("A file in the tar archive could not be completely read.");
        }

        return bytes;
    }

    
    public static void main(String[] args) {
        if (args.length != 2) {
            exitWithHelp();
        }

        try {
            TarToSeqFile me = new TarToSeqFile();
            me.setInput(new File(args[0]));
            me.setOutput(new File(args[1]));
            me.execute();
        } catch (Exception e) {
            e.printStackTrace();
            exitWithHelp();
        }
    }

    public static void exitWithHelp() {
        System.err.println("Usage:  <tarfile> TarToSeqFile  <output>\n\n" +
                           "<tarfile> may be GZIP or BZIP2 compressed, must have a\n" +
                           "recognizable extension .tar, .tar.gz, .tgz, .tar.bz2, or .tbz2.");
        System.exit(1);
    }
}

In this way we can write files to a single Sequence file, to test it further we will read from the Sequence file and list the keys of the file as output
import org.apache.hadoop.conf.Configuration;
import org.apache.hadoop.fs.FileSystem;
import org.apache.hadoop.fs.LocalFileSystem;
import org.apache.hadoop.fs.Path;
import org.apache.hadoop.io.SequenceFile;
import org.apache.hadoop.io.Writable;


public class SeqKeyList {

    private String inputFile;
    private LocalSetup setup;

    public SeqKeyList() throws Exception {
        setup = new LocalSetup();
    }

    public void setInput(String filename) {
        inputFile = filename;
    }

    
    public void execute() throws Exception {
        Path path = new Path(inputFile);
        SequenceFile.Reader reader = 
            new SequenceFile.Reader(setup.getLocalFileSystem(), path, setup.getConf());

        try {
            System.err.println("Key type is " + reader.getKeyClassName());
            System.err.println("Value type is " + reader.getValueClassName());
            if (reader.isCompressed()) {
                System.err.println("Values are compressed.");
            }
            if (reader.isBlockCompressed()) {
                System.err.println("Records are block-compressed.");
            }
            System.err.println("Compression type is " + reader.getCompressionCodec().getClass().getName());
            System.err.println("");

            Writable key = (Writable)(reader.getKeyClass().newInstance());
            while (reader.next(key)) {
                System.out.println(key.toString());
            }
        } finally {
            reader.close();
        }
    }

    public static void main(String[] args) {
        if (args.length != 1) {
            exitWithHelp();
        }

        try {
            SeqKeyList me = new SeqKeyList();
            me.setInput(args[0]);
            me.execute();
        } catch (Exception e) {
            e.printStackTrace();
            exitWithHelp();
        }
    }

    
    public static void exitWithHelp() {
        System.err.println("Usage: SeqKeyList   <sequence-file>\n" +
                           "Prints a list of keys in the sequence file, one per line.");
        System.exit(1);
    }
}


Hadoop : How to read and write a Map file

MapFile
A MapFile is a sorted SequenceFile with an index to permit lookups by key. MapFile can be thought of as a persistent form of java.util.Map (although it doesn’t implement this interface), which is able to grow beyond the size of a Map that is kept in memory.

Writing a MapFile
Writing a MapFile is similar to writing a SequenceFile: you create an instance of MapFile.Writer, then call the append() method to add entries in order. (Attempting to add entries out of order will result in an IOException.) Keys must be instances of WritableComparable, and values must be Writable—contrast this to SequenceFile, which can use any serialization framework for its entries.

Iterating through the entries in order in a MapFile is similar to the procedure for a SequenceFile: you create a MapFile.Reader, then call the next() method until it returns false, signifying that no entry was read because the end of the file was reached

our data sample:
#custId orderNo
965412 S986512
965413 S986513
965414 S986514
965415 S986515
965416 S986516

Now configure your dependencies in pom.xml
<project xmlns="http://maven.apache.org/POM/4.0.0" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"
  xsi:schemaLocation="http://maven.apache.org/POM/4.0.0 http://maven.apache.org/maven-v4_0_0.xsd">
  <modelVersion>4.0.0</modelVersion>
  <groupId>com.rjkrsinghhadoop</groupId>
  <artifactId>MapFileConverter</artifactId>
  <packaging>jar</packaging>
  <version>1.0</version>
  <name>MapFileConverter</name>
  <url>http://maven.apache.org</url>
  <dependencies>
        <dependency>
            <groupId>junit</groupId>
            <artifactId>junit</artifactId>
            <version>4.7</version>
            <scope>test</scope>
        </dependency>
        <dependency>
            <groupId>org.apache.hadoop</groupId>
            <artifactId>hadoop-core</artifactId>
            <version>1.0.4</version>
        </dependency>
        <dependency>
            <groupId>commons-logging</groupId>
            <artifactId>commons-logging-api</artifactId>
            <version>1.0.4</version>
        </dependency>
        <dependency>
            <groupId>commons-logging</groupId>
            <artifactId>commons-logging</artifactId>
            <version>1.0.4</version>
            <scope>compile</scope>
        </dependency>
        <dependency>
            <groupId>commons-cli</groupId>
            <artifactId>commons-cli</artifactId>
            <version>1.2</version>
        </dependency>
    </dependencies>

<!--
    <repositories>
        <repository>
            <id>libdir</id>
            <url>file://${basedir}/lib</url>
        </repository>
    </repositories>
-->

    <build>
        <finalName>exploringhadoop</finalName>
        <plugins>
                        <plugin>
                                <groupId>org.apache.maven.plugins</groupId>
                                <artifactId>maven-compiler-plugin</artifactId>
                                <configuration>
                                        <source>1.6</source>
                                        <target>1.6</target>
                                </configuration>
                        </plugin>
                        <plugin>
                                <artifactId>maven-assembly-plugin</artifactId>
                                <configuration>
                                        <finalName>${project.name}-${project.version}</finalName>
                                        <appendAssemblyId>true</appendAssemblyId>
                                        <descriptors>
                                                <descriptor>src/main/assembly/assembly.xml</descriptor>
                                        </descriptors>
                                </configuration>
                        </plugin>
        </plugins>
    </build>
</project>

Create MapFileConverter.java which will responsible for converting the text file to the Map file
package com.rjkrsinghhadoop;


import  java.io.IOException;
import org.apache.hadoop.io.Text;
import org.apache.hadoop.io.MapFile;
import org.apache.hadoop.conf.Configuration;
import org.apache.hadoop.fs.FileSystem;
import org.apache.hadoop.fs.Path;
import org.apache.hadoop.io.IOUtils;
import org.apache.hadoop.fs.FSDataInputStream;

public class MapFileConverter {

  @SuppressWarnings("deprecation")
        public static void main(String[] args) throws IOException{
                
                Configuration conf = new Configuration();
                FileSystem fs;
                
                try {
                        fs = FileSystem.get(conf);
                        
                        Path inputFile = new Path(args[0]);
                        Path outputFile = new Path(args[1]);

                      Text txtKey = new Text();
                          Text txtValue = new Text();

                        String strLineInInputFile = "";
                        String lstKeyValuePair[] = null;
                        MapFile.Writer writer = null;
                        
                        FSDataInputStream inputStream = fs.open(inputFile);

                        try {
                                writer = new MapFile.Writer(conf, fs, outputFile.toString(),
                                                txtKey.getClass(), txtKey.getClass());
                                writer.setIndexInterval(1);
                                while (inputStream.available() > 0) {
                                        strLineInInputFile = inputStream.readLine();
                                        lstKeyValuePair = strLineInInputFile.split("\\t");
                                        txtKey.set(lstKeyValuePair[0]);
                                        txtValue.set(lstKeyValuePair[1]);
                                        writer.append(txtKey, txtValue);
                                }
                        } finally {
                                IOUtils.closeStream(writer);
                                System.out.println("Map file created successfully!!");
                  }
        } catch (IOException e) {
                        e.printStackTrace();
                }        
        }
}

to look up a file based on the provided key we will use MapFileReader.

package com.rjkrsinghhadoop;


import java.io.IOException;
import org.apache.hadoop.fs.FileSystem;
import org.apache.hadoop.io.MapFile;
import org.apache.hadoop.io.Text;
import org.apache.hadoop.conf.Configuration;

public class MapFileReader {

  
        @SuppressWarnings("deprecation")
        public static void main(String[] args) throws IOException {

                Configuration conf = new Configuration();
                FileSystem fs = null;
                Text txtKey = new Text(args[1]);
                Text txtValue = new Text();
                MapFile.Reader reader = null;

                try {
                        fs = FileSystem.get(conf);

                        try {
                                reader = new MapFile.Reader(fs, args[0].toString(), conf);
                                reader.get(txtKey, txtValue);
                        } catch (IOException e) {
                                e.printStackTrace();
                        }

                } catch (IOException e) {
                        e.printStackTrace();
                }
                System.out.println("The value for Key "+ txtKey.toString() +" is "+ txtValue.toString());
        }
}

to ship your code in the jar file we will need an assembly descriptor create a assembly.xml in the resources folder as follows:
<assembly
    xmlns="http://maven.apache.org/plugins/maven-assembly-plugin/assembly/1.1.0"
    xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"
    xsi:schemaLocation="http://maven.apache.org/plugins/maven-assembly-plugin/assembly/1.1.0 http://maven.apache.org/xsd/assembly-1.1.0.xsd">
    <id>job</id>
    <formats>
        <format>jar</format>
    </formats>
    <includeBaseDirectory>false</includeBaseDirectory>
    <dependencySets>
        <dependencySet>
            <unpack>false</unpack>
            <scope>runtime</scope>
            <outputDirectory>lib</outputDirectory>
            <excludes>
                <exclude>${artifact.groupId}:${artifact.artifactId}</exclude>
            </excludes>
        </dependencySet>
        <dependencySet>
            <unpack>false</unpack>
            <scope>system</scope>
            <outputDirectory>lib</outputDirectory>
            <excludes>
                <exclude>${artifact.groupId}:${artifact.artifactId}</exclude>
            </excludes>
        </dependencySet>
    </dependencySets>
    <fileSets>
        <fileSet>
            <directory>${basedir}/target/classes</directory>
            <outputDirectory>/</outputDirectory>
            <excludes>
                <exclude>*.jar</exclude>
            </excludes>
        </fileSet>
    </fileSets>
</assembly>

now run mvn assembly:assembly which will create a jar file in the target directory, which is ready to be run on your hadoop cluster.