对;问题是,这是不是只是 protobuf的 - 这是一个混合的文件格式(即defined here包括内部各种格式之间的protobuf它还集成了压缩(虽然这看起来是可选)
我。我从规范中拉开了我所能做的,我在这里有一个C#阅读器,它使用protobuf-net处理这些块 - 它愉快地通读该文件到最后 - 我可以告诉你有4515个块(BlockHeader
) 。当它到达Blob
时,我有点困惑,关于如何规格OSMHeader
和OSMData
- 我接受建议在这里!我也用ZLIB.NET来处理正在使用的zlib压缩。的为了解决这个问题,我已经解决了处理ZLIB数据并根据声称的大小对其进行验证,以检查它是否至少是理智的。
如果你能弄清楚(或询问作者)他们是如何分离的OSMHeader
和OSMData
我会愉快地开始其他事情。我希望你不要介意我已经停在这里 - 但它一直是几个小时内,p
using System;
using System.IO;
using OpenStreetMap; // where my .proto-generated entities are living
using ProtoBuf; // protobuf-net
using zlib; // ZLIB.NET
class OpenStreetMapParser
{
static void Main()
{
using (var file = File.OpenRead("us-northeast.osm.pbf"))
{
// from http://wiki.openstreetmap.org/wiki/ProtocolBufBinary:
//A file contains a header followed by a sequence of fileblocks. The design is intended to allow future random-access to the contents of the file and skipping past not-understood or unwanted data.
//The format is a repeating sequence of:
//int4: length of the BlockHeader message in network byte order
//serialized BlockHeader message
//serialized Blob message (size is given in the header)
int length, blockCount = 0;
while (Serializer.TryReadLengthPrefix(file, PrefixStyle.Fixed32, out length))
{
// I'm just being lazy and re-using something "close enough" here
// note that v2 has a big-endian option, but Fixed32 assumes little-endian - we
// actually need the other way around (network byte order):
uint len = (uint)length;
len = ((len & 0xFF) << 24) | ((len & 0xFF00) << 8) | ((len & 0xFF0000) >> 8) | ((len & 0xFF000000) >> 24);
length = (int)len;
BlockHeader header;
// again, v2 has capped-streams built in, but I'm deliberately
// limiting myself to v1 features
using (var tmp = new LimitedStream(file, length))
{
header = Serializer.Deserialize<BlockHeader>(tmp);
}
Blob blob;
using (var tmp = new LimitedStream(file, header.datasize))
{
blob = Serializer.Deserialize<Blob>(tmp);
}
if(blob.zlib_data == null) throw new NotSupportedException("I'm only handling zlib here!");
using(var ms = new MemoryStream(blob.zlib_data))
using(var zlib = new ZLibStream(ms))
{ // at this point I'm very unclear how the OSMHeader and OSMData are packed - it isn't clear
// read this to the end, to check we can parse the zlib
int payloadLen = 0;
while (zlib.ReadByte() >= 0) payloadLen++;
if (payloadLen != blob.raw_size) throw new FormatException("Screwed that up...");
}
blockCount++;
Console.WriteLine("Read block " + blockCount.ToString());
}
Console.WriteLine("all done");
Console.ReadLine();
}
}
}
abstract class InputStream : Stream
{
protected abstract int ReadNextBlock(byte[] buffer, int offset, int count);
public sealed override int Read(byte[] buffer, int offset, int count)
{
int bytesRead, totalRead = 0;
while (count > 0 && (bytesRead = ReadNextBlock(buffer, offset, count)) > 0)
{
count -= bytesRead;
offset += bytesRead;
totalRead += bytesRead;
pos += bytesRead;
}
return totalRead;
}
long pos;
public override void Write(byte[] buffer, int offset, int count)
{
throw new NotImplementedException();
}
public override void SetLength(long value)
{
throw new NotImplementedException();
}
public override long Position
{
get
{
return pos;
}
set
{
if (pos != value) throw new NotImplementedException();
}
}
public override long Length
{
get { throw new NotImplementedException(); }
}
public override void Flush()
{
throw new NotImplementedException();
}
public override bool CanWrite
{
get { return false; }
}
public override bool CanRead
{
get { return true; }
}
public override bool CanSeek
{
get { return false; }
}
public override long Seek(long offset, SeekOrigin origin)
{
throw new NotImplementedException();
}
}
class ZLibStream : InputStream
{ // uses ZLIB.NET: http://www.componentace.com/download/download.php?editionid=25
private ZInputStream reader; // seriously, why isn't this a stream?
public ZLibStream(Stream stream)
{
reader = new ZInputStream(stream);
}
public override void Close()
{
reader.Close();
base.Close();
}
protected override int ReadNextBlock(byte[] buffer, int offset, int count)
{
// OMG! reader.Read is the base-stream, reader.read is decompressed! yeuch
return reader.read(buffer, offset, count);
}
}
// deliberately doesn't dispose the base-stream
class LimitedStream : InputStream
{
private Stream stream;
private long remaining;
public LimitedStream(Stream stream, long length)
{
if (length < 0) throw new ArgumentOutOfRangeException("length");
if (stream == null) throw new ArgumentNullException("stream");
if (!stream.CanRead) throw new ArgumentException("stream");
this.stream = stream;
this.remaining = length;
}
protected override int ReadNextBlock(byte[] buffer, int offset, int count)
{
if(count > remaining) count = (int)remaining;
int bytesRead = stream.Read(buffer, offset, count);
if (bytesRead > 0) remaining -= bytesRead;
return bytesRead;
}
}
我protobuf网的作者;我现在正在“工作”的时间,但我今天晚些时候会尝试着看这个问题,看看有什么问题 – 2011-01-13 07:32:48
我知道你是谁,我下载了你的软件。我喜欢括号中的作品哈哈。感谢您的帮助(和框架)! – jonperl 2011-01-13 18:18:49