* Added new (currently private) API class BMimeSnifferAddon,

representing the interface for, well, MIME sniffer add-ons.
* Implemented the respective add-on manager and make use of it in
  the MIME database code. Unfortunately the MIME DB code completely
  lives in libbe.so and hence I had to put my code there too.
  IMHO we should (one day) remove the direct (read-only) MIME DB
  access from libbe and move everything into the registrar.
  Currently the add-on manager supports built-in add-ons only; it
  doesn't really load anything from disk ATM.
* Added a built-in text sniffer add-on to the registrar. It's based
  upon the BSD file tool code.

This closes bug #250 (plain text files are identified as such, now).



git-svn-id: file:///srv/svn/repos/haiku/haiku/trunk@17784 a95241bf-73f2-0310-859d-f6bbb57e9c96
This commit is contained in:
Ingo Weinhold
2006-06-09 20:48:50 +00:00
parent 642090fdc6
commit e4f35acf7c
14 changed files with 1330 additions and 25 deletions
+4 -2
View File
@@ -341,7 +341,8 @@ rule UseLegacyObjectHeaders
rule FStandardOSHeaders
{
local osIncludes = add-ons add-ons/file_system add-ons/graphics
add-ons/input_server add-ons/screen_saver
add-ons/input_server add-ons/registrar
add-ons/screen_saver
add-ons/tracker app device drivers game interface
kernel media mail midi midi2 net opengl storage support
translation ;
@@ -352,7 +353,8 @@ rule FStandardOSHeaders
rule FStandardHeaders
{
local osIncludes = add-ons add-ons/file_system add-ons/graphics
add-ons/input_server add-ons/screen_saver
add-ons/input_server add-ons/registrar
add-ons/screen_saver
add-ons/tracker app device drivers game interface
kernel media mail midi midi2 net opengl storage support
translation ;
@@ -0,0 +1,32 @@
/* MIME Sniffer Add-On Protocol
*
* Copyright 2006, Haiku Inc. All Rights Reserved.
* Distributed under the terms of the MIT License.
*/
#ifndef _MIME_SNIFFER_ADDON_H
#define _MIME_SNIFFER_ADDON_H
#include <SupportDefs.h>
class BFile;
class BMimeType;
// **********************************
// *** WARNING: EXPERIMENTAL API! ***
// **********************************
class BMimeSnifferAddon {
public:
BMimeSnifferAddon();
virtual ~BMimeSnifferAddon();
virtual size_t MinimalBufferSize();
virtual float GuessMimeType(const char* fileName,
BMimeType* type);
virtual float GuessMimeType(BFile* file,
const void* buffer, int32 length,
BMimeType* type);
};
#endif // _MIME_SNIFFER_ADDON_H
@@ -0,0 +1,62 @@
/*
* Copyright 2006, Ingo Weinhold <[email protected]>.
* All rights reserved. Distributed under the terms of the MIT License.
*/
#ifndef MIME_SNIFFER_ADDON_MANAGER_H
#define MIME_SNIFFER_ADDON_MANAGER_H
#include <List.h>
#include <Locker.h>
class BFile;
class BMimeSnifferAddon;
class BMimeType;
namespace BPrivate {
namespace Storage {
namespace Mime {
class MimeSnifferAddonManager {
private:
MimeSnifferAddonManager();
~MimeSnifferAddonManager();
public:
static MimeSnifferAddonManager* Default();
static status_t CreateDefault();
static void DeleteDefault();
status_t AddMimeSnifferAddon(BMimeSnifferAddon* addon);
size_t MinimalBufferSize();
float GuessMimeType(const char* fileName,
BMimeType* type);
float GuessMimeType(BFile* file,
const void* buffer, int32 length,
BMimeType* type);
private:
struct AddonReference;
status_t _GetAddons(AddonReference**& references,
int32& count);
void _PutAddons(AddonReference** references,
int32 count);
static MimeSnifferAddonManager* sManager;
BLocker fLock;
BList fAddons;
size_t fMinimalBufferSize;
};
} // namespace Mime
} // namespace Storage
} // namespace BPrivate
using BPrivate::Storage::Mime::MimeSnifferAddonManager;
#endif // MIME_SNIFFER_ADDON_MANAGER_H
+2 -1
View File
@@ -50,7 +50,8 @@ public:
};
private:
status_t BuildRuleList();
status_t GuessMimeType(BPositionIO *data, BString *type);
status_t GuessMimeType(BFile* file, const void *buffer, int32 length,
BString *type);
ssize_t MaxBytesNeeded();
status_t ProcessType(const char *type, ssize_t *bytesNeeded);
+2
View File
@@ -41,6 +41,8 @@ MergeObject <libbe>storage_kit.o :
CreateAppMetaMimeThread.cpp
Database.cpp
InstalledTypes.cpp
MimeSnifferAddon.cpp
MimeSnifferAddonManager.cpp
MimeUpdateThread.cpp
SnifferRules.cpp
Supertype.cpp
+15
View File
@@ -16,6 +16,7 @@
#include <Path.h>
#include <String.h>
#include <mime/database_support.h>
#include <mime/MimeSnifferAddonManager.h>
#include <storage_support.h>
#include <new>
@@ -100,6 +101,20 @@ AssociatedTypes::GuessMimeType(const char *filename, BString *result)
status_t err = filename && result ? B_OK : B_BAD_VALUE;
if (!err && !fHaveDoneFullBuild)
err = BuildAssociatedTypesTable();
// if we have an MimeSnifferAddonManager, let's give it a shot first
if (!err) {
MimeSnifferAddonManager* manager = MimeSnifferAddonManager::Default();
if (manager) {
BMimeType mimeType;
float priority = manager->GuessMimeType(filename, &mimeType);
if (priority >= 0) {
*result = mimeType.Type();
return B_OK;
}
}
}
if (!err) {
// Extract the extension from the file
const char *rawExtension = strrchr(filename, '.');
@@ -0,0 +1,38 @@
/*
* Copyright 2006, Ingo Weinhold <[email protected]>.
* All rights reserved. Distributed under the terms of the MIT License.
*/
#include <MimeSnifferAddon.h>
// constructor
BMimeSnifferAddon::BMimeSnifferAddon()
{
}
// destructor
BMimeSnifferAddon::~BMimeSnifferAddon()
{
}
// MinimalBufferSize
size_t
BMimeSnifferAddon::MinimalBufferSize()
{
return 0;
}
// GuessMimeType
float
BMimeSnifferAddon::GuessMimeType(const char* fileName, BMimeType* type)
{
return -1;
}
// GuessMimeType
float
BMimeSnifferAddon::GuessMimeType(BFile* file, const void* buffer, int32 length,
BMimeType* type)
{
return -1;
}
@@ -0,0 +1,224 @@
/*
* Copyright 2006, Ingo Weinhold <[email protected]>.
* All rights reserved. Distributed under the terms of the MIT License.
*/
#include <mime/MimeSnifferAddonManager.h>
#include <new>
#include <Autolock.h>
#include <MimeSnifferAddon.h>
#include <MimeType.h>
using std::nothrow;
// singleton instance
MimeSnifferAddonManager* MimeSnifferAddonManager::sManager = NULL;
// AddonReference
struct MimeSnifferAddonManager::AddonReference {
AddonReference(BMimeSnifferAddon* addon)
: fAddon(addon),
fReferenceCount(1)
{
}
~AddonReference()
{
delete fAddon;
}
BMimeSnifferAddon* Addon() const
{
return fAddon;
}
void GetReference()
{
atomic_add(&fReferenceCount, 1);
}
void PutReference()
{
if (atomic_add(&fReferenceCount, -1) == 1)
delete this;
}
private:
BMimeSnifferAddon* fAddon;
vint32 fReferenceCount;
};
// constructor
MimeSnifferAddonManager::MimeSnifferAddonManager()
: fLock("mime sniffer manager"),
fAddons(20),
fMinimalBufferSize(0)
{
}
// destructor
MimeSnifferAddonManager::~MimeSnifferAddonManager()
{
}
// Default
MimeSnifferAddonManager*
MimeSnifferAddonManager::Default()
{
return sManager;
}
// CreateDefault
status_t
MimeSnifferAddonManager::CreateDefault()
{
MimeSnifferAddonManager* manager = new(nothrow) MimeSnifferAddonManager;
if (!manager)
return B_NO_MEMORY;
sManager = manager;
return B_OK;
}
// DeleteDefault
void
MimeSnifferAddonManager::DeleteDefault()
{
MimeSnifferAddonManager* manager = sManager;
sManager = NULL;
delete manager;
}
// AddMimeSnifferAddon
status_t
MimeSnifferAddonManager::AddMimeSnifferAddon(BMimeSnifferAddon* addon)
{
if (!addon)
return B_BAD_VALUE;
BAutolock locker(fLock);
if (!locker.IsLocked())
return B_ERROR;
// create a reference for the addon
AddonReference* reference = new(nothrow) AddonReference(addon);
if (!reference)
return B_NO_MEMORY;
// add the reference
if (!fAddons.AddItem(reference)) {
delete reference;
return B_NO_MEMORY;
}
// update minimal buffer size
size_t minBufferSize = addon->MinimalBufferSize();
if (minBufferSize > fMinimalBufferSize)
fMinimalBufferSize = minBufferSize;
return B_OK;
}
// MinimalBufferSize
size_t
MimeSnifferAddonManager::MinimalBufferSize()
{
return fMinimalBufferSize;
}
// GuessMimeType
float
MimeSnifferAddonManager::GuessMimeType(const char* fileName, BMimeType* type)
{
// get addons
AddonReference** addons = NULL;
int32 count = 0;
status_t error = _GetAddons(addons, count);
if (error != B_OK)
return -1;
// iterate over the addons and find the most fitting type
float bestPriority = -1;
for (int32 i = 0; i < count; i++) {
BMimeType currentType;
float priority = addons[i]->Addon()->GuessMimeType(fileName,
&currentType);
if (priority > bestPriority) {
type->SetTo(currentType.Type());
bestPriority = priority;
}
}
// release addons
_PutAddons(addons, count);
return bestPriority;
}
// GuessMimeType
float
MimeSnifferAddonManager::GuessMimeType(BFile* file, const void* buffer,
int32 length, BMimeType* type)
{
// get addons
AddonReference** addons = NULL;
int32 count = 0;
status_t error = _GetAddons(addons, count);
if (error != B_OK)
return -1;
// iterate over the addons and find the most fitting type
float bestPriority = -1;
for (int32 i = 0; i < count; i++) {
BMimeType currentType;
float priority = addons[i]->Addon()->GuessMimeType(file, buffer,
length, &currentType);
if (priority > bestPriority) {
type->SetTo(currentType.Type());
bestPriority = priority;
}
}
// release addons
_PutAddons(addons, count);
return bestPriority;
}
// _GetAddons
status_t
MimeSnifferAddonManager::_GetAddons(AddonReference**& references, int32& count)
{
BAutolock locker(fLock);
if (!locker.IsLocked())
return B_ERROR;
count = fAddons.CountItems();
references = new(nothrow) AddonReference*[count];
if (!references)
return B_NO_MEMORY;
for (int32 i = 0; i < count; i++) {
references[i] = (AddonReference*)fAddons.ItemAt(i);
references[i]->GetReference();
}
return B_OK;
}
// _PutAddons
void
MimeSnifferAddonManager::_PutAddons(AddonReference** references, int32 count)
{
for (int32 i = 0; i < count; i++)
references[i]->PutReference();
delete[] references;
}
+55 -21
View File
@@ -14,6 +14,7 @@
#include <File.h>
#include <MimeType.h>
#include <mime/database_support.h>
#include <mime/MimeSnifferAddonManager.h>
#include <sniffer/Parser.h>
#include <sniffer/Rule.h>
#include <StorageDefs.h>
@@ -129,8 +130,8 @@ SnifferRules::~SnifferRules()
/*! \brief Guesses a MIME type for the supplied entry_ref.
Only the data in the given entry is considered, not the filename or
its extension. Please see GuessMimeType(BPositionIO*, BString*) for
more details.
its extension. Please see GuessMimeType(BFile*, const void *, int32,
BString*) for more details.
\param ref The entry to sniff
\param type Pointer to a pre-allocated BString which is set to the
@@ -173,10 +174,8 @@ SnifferRules::GuessMimeType(const entry_ref *ref, BString *type)
}
// Now sniff the buffer
if (!err) {
BMemoryIO data(buffer, bytes);
err = GuessMimeType(&data, type);
}
if (!err)
err = GuessMimeType(&file, buffer, bytes, type);
return err;
}
@@ -184,7 +183,8 @@ SnifferRules::GuessMimeType(const entry_ref *ref, BString *type)
// GuessMimeType
/*! \brief Guesses a MIME type for the given chunk of data.
Please see GuessMimeType(BPositionIO*, BString*) for more details.
Please see GuessMimeType(BFile*, const void *, int32, BString*) for more
details.
\param buffer Pointer to a data buffer to sniff
\param length The length of the data buffer pointed to by \a buffer
@@ -198,15 +198,7 @@ SnifferRules::GuessMimeType(const entry_ref *ref, BString *type)
status_t
SnifferRules::GuessMimeType(const void *buffer, int32 length, BString *type)
{
status_t err = buffer && type ? B_OK : B_BAD_VALUE;
// Wrap a BMemoryIO around the buffer and call our private
// GuessMimeType(BPositionIO*, BString*) function to do the
// dirty work
if (!err) {
BMemoryIO data(buffer, length);
err = GuessMimeType(&data, type);
}
return err;
return GuessMimeType(NULL, buffer, length, type);
}
// SetSnifferRule
@@ -422,7 +414,9 @@ SnifferRules::BuildRuleList()
"supertype/subtype" form rules are checked before "supertype-only" form
rules if their priorities happen to be identical).
\param data The data to sniff
\param file The file to sniff. May be \c NULL. \a buffer is always given.
\param buffer Pointer to a data buffer to sniff
\param length The length of the data buffer pointed to by \a buffer
\param type Pointer to a pre-allocated BString which is set to the
resulting MIME type.
\return
@@ -431,11 +425,30 @@ SnifferRules::BuildRuleList()
- error code: failure
*/
status_t
SnifferRules::GuessMimeType(BPositionIO *data, BString *type)
SnifferRules::GuessMimeType(BFile* file, const void *buffer, int32 length,
BString *type)
{
status_t err = data && type ? B_OK : B_BAD_VALUE;
status_t err = buffer && type ? B_OK : B_BAD_VALUE;
if (err)
return err;
// wrap the buffer by a BMemoryIO
BMemoryIO data(buffer, length);
if (!err && !fHaveDoneFullBuild)
err = BuildRuleList();
// first ask the MimeSnifferAddonManager for a suitable type
float addonPriority = -1;
BMimeType mimeType;
if (!err) {
MimeSnifferAddonManager* manager = MimeSnifferAddonManager::Default();
if (manager) {
addonPriority = manager->GuessMimeType(file, buffer, length,
&mimeType);
}
}
if (!err) {
// Run through our rule list, which is sorted in order of
// descreasing priority, and see if one of the rules sniffs
@@ -445,7 +458,15 @@ SnifferRules::GuessMimeType(BPositionIO *data, BString *type)
i++)
{
if (i->rule) {
if (i->rule->Sniff(data)) {
// If an add-on identified the type with a priority at least
// as great as the remaining rules, we can stop further
// processing and return the type found by the add-on.
if (i->rule->Priority() <= addonPriority) {
*type = mimeType.Type();
return B_OK;
}
if (i->rule->Sniff(&data)) {
type->SetTo(i->type.c_str());
return B_OK;
}
@@ -456,6 +477,13 @@ SnifferRules::GuessMimeType(BPositionIO *data, BString *type)
i->type.c_str(), i->rule_string.c_str()));
}
}
// The sniffer add-on manager might have returned a low priority
// (lower than any of a rule).
if (addonPriority >= 0) {
*type = mimeType.Type();
return B_OK;
}
// If we get here, we didn't find a damn thing
err = kMimeGuessFailureError;
@@ -479,8 +507,14 @@ ssize_t
SnifferRules::MaxBytesNeeded()
{
ssize_t err = fHaveDoneFullBuild ? B_OK : BuildRuleList();
if (!err)
if (!err) {
err = fMaxBytesNeeded;
MimeSnifferAddonManager* manager = MimeSnifferAddonManager::Default();
if (manager) {
fMaxBytesNeeded = max(fMaxBytesNeeded,
(ssize_t)manager->MinimalBufferSize());
}
}
return err;
}
+1
View File
@@ -29,6 +29,7 @@ Server registrar
RosterAppInfo.cpp
RosterSettingsCharStream.cpp
ShutdownProcess.cpp
TextSnifferAddon.cpp
TRoster.cpp
Watcher.cpp
WatchingService.cpp
+11 -1
View File
@@ -4,8 +4,9 @@
#include <ClassInfo.h>
#include <Message.h>
#include <Messenger.h>
#include <mime/UpdateMimeInfoThread.h>
#include <mime/CreateAppMetaMimeThread.h>
#include <mime/MimeSnifferAddonManager.h>
#include <mime/UpdateMimeInfoThread.h>
#include <Path.h>
#include <RegistrarDefs.h>
#include <String.h>
@@ -18,6 +19,7 @@ using namespace std;
using namespace BPrivate;
#include "MIMEManager.h"
#include "TextSnifferAddon.h"
/*!
\class MIMEManager
@@ -35,6 +37,14 @@ MIMEManager::MIMEManager()
, fThreadManager()
{
AddHandler(&fThreadManager);
// prepare the MimeSnifferAddonManager and the built-in add-ons
status_t error = MimeSnifferAddonManager::CreateDefault();
if (error == B_OK) {
MimeSnifferAddonManager* addonManager
= MimeSnifferAddonManager::Default();
addonManager->AddMimeSnifferAddon(new(nothrow) TextSnifferAddon());
}
}
// destructor
+674
View File
@@ -0,0 +1,674 @@
/*
* Copyright 2006, Ingo Weinhold <bonefish@cs.tu-berlin.de>.
* All rights reserved. Distributed under the terms of the MIT License.
*/
#include "TextSnifferAddon.h"
#include <MimeType.h>
static int file_ascmagic(const unsigned char *buf, size_t nbytes,
BMimeType* mimeType);
// constructor
TextSnifferAddon::TextSnifferAddon()
{
}
// destructor
TextSnifferAddon::~TextSnifferAddon()
{
}
// MinimalBufferSize
size_t
TextSnifferAddon::MinimalBufferSize()
{
return 512;
}
// GuessMimeType
float
TextSnifferAddon::GuessMimeType(const char* fileName, BMimeType* type)
{
// we check content only
return -1;
}
// GuessMimeType
float
TextSnifferAddon::GuessMimeType(BFile* file, const void* buffer, int32 length,
BMimeType* type)
{
if (file_ascmagic((const unsigned char*)buffer, length, type)) {
// If the buffer is very short, we return a lower priority. Maybe
// someone else knows better.
if (length < 20)
return .0f;
return 0.25f;
}
return -1;
}
// #pragma mark - ascmagic.c from the BSD file tool
/*
* The following code has been taken from version 4.17 of the BSD file tool,
* file ascmagic.c, modified for our purpose.
*/
/*
* Copyright (c) Ian F. Darwin 1986-1995.
* Software written by Ian F. Darwin and others;
* maintained 1995-present by Christos Zoulas and others.
*
* Redistribution and use in source and binary forms, with or without
* modification, are permitted provided that the following conditions
* are met:
* 1. Redistributions of source code must retain the above copyright
* notice immediately at the beginning of the file, without modification,
* this list of conditions, and the following disclaimer.
* 2. Redistributions in binary form must reproduce the above copyright
* notice, this list of conditions and the following disclaimer in the
* documentation and/or other materials provided with the distribution.
*
* THIS SOFTWARE IS PROVIDED BY THE AUTHOR AND CONTRIBUTORS ``AS IS'' AND
* ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
* ARE DISCLAIMED. IN NO EVENT SHALL THE AUTHOR OR CONTRIBUTORS BE LIABLE FOR
* ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
* OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
* HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
* LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
* OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
* SUCH DAMAGE.
*/
/*
* ASCII magic -- file types that we know based on keywords
* that can appear anywhere in the file.
*
* Extensively modified by Eric Fischer <enf@pobox.com> in July, 2000,
* to handle character codes other than ASCII on a unified basis.
*
* Joerg Wunsch <joerg@freebsd.org> wrote the original support for 8-bit
* international characters, now subsumed into this file.
*/
#include <stdio.h>
#include <string.h>
#include <memory.h>
#include <ctype.h>
#include <stdlib.h>
#include <unistd.h>
#include "names.h"
typedef unsigned long my_unichar;
#define MAXLINELEN 300 /* longest sane line length */
#define ISSPC(x) ((x) == ' ' || (x) == '\t' || (x) == '\r' || (x) == '\n' \
|| (x) == 0x85 || (x) == '\f')
static int looks_ascii(const unsigned char *, size_t, my_unichar *, size_t *);
static int looks_utf8(const unsigned char *, size_t, my_unichar *, size_t *);
static int looks_unicode(const unsigned char *, size_t, my_unichar *, size_t *);
static int looks_latin1(const unsigned char *, size_t, my_unichar *, size_t *);
static int looks_extended(const unsigned char *, size_t, my_unichar *, size_t *);
static void from_ebcdic(const unsigned char *, size_t, unsigned char *);
static int ascmatch(const unsigned char *, const my_unichar *, size_t);
static int
file_ascmagic(const unsigned char *buf, size_t nbytes, BMimeType* mimeType)
{
size_t i;
unsigned char *nbuf = NULL;
my_unichar *ubuf = NULL;
size_t ulen;
struct names *p;
int rv = -1;
const char *code = NULL;
const char *code_mime = NULL;
const char *type = NULL;
const char *subtype = NULL;
const char *subtype_mime = NULL;
int has_escapes = 0;
int has_backspace = 0;
int seen_cr = 0;
int n_crlf = 0;
int n_lf = 0;
int n_cr = 0;
int n_nel = 0;
int last_line_end = -1;
int has_long_lines = 0;
if ((nbuf = (unsigned char*)malloc((nbytes + 1) * sizeof(nbuf[0]))) == NULL)
goto done;
if ((ubuf = (my_unichar*)malloc((nbytes + 1) * sizeof(ubuf[0]))) == NULL)
goto done;
/*
* Then try to determine whether it's any character code we can
* identify. Each of these tests, if it succeeds, will leave
* the text converted into one-my_unichar-per-character Unicode in
* ubuf, and the number of characters converted in ulen.
*/
if (looks_ascii(buf, nbytes, ubuf, &ulen)) {
code = "ASCII";
code_mime = "us-ascii";
type = "text";
} else if (looks_utf8(buf, nbytes, ubuf, &ulen)) {
code = "UTF-8 Unicode";
code_mime = "utf-8";
type = "text";
} else if ((i = looks_unicode(buf, nbytes, ubuf, &ulen)) != 0) {
if (i == 1)
code = "Little-endian UTF-16 Unicode";
else
code = "Big-endian UTF-16 Unicode";
type = "character data";
code_mime = "utf-16"; /* is this defined? */
} else if (looks_latin1(buf, nbytes, ubuf, &ulen)) {
code = "ISO-8859";
type = "text";
code_mime = "iso-8859-1";
} else if (looks_extended(buf, nbytes, ubuf, &ulen)) {
code = "Non-ISO extended-ASCII";
type = "text";
code_mime = "unknown";
} else {
from_ebcdic(buf, nbytes, nbuf);
if (looks_ascii(nbuf, nbytes, ubuf, &ulen)) {
code = "EBCDIC";
type = "character data";
code_mime = "ebcdic";
} else if (looks_latin1(nbuf, nbytes, ubuf, &ulen)) {
code = "International EBCDIC";
type = "character data";
code_mime = "ebcdic";
} else {
rv = 0;
goto done; /* doesn't look like text at all */
}
}
if (nbytes <= 1) {
rv = 0;
goto done;
}
/*
* for troff, look for . + letter + letter or .\";
* this must be done to disambiguate tar archives' ./file
* and other trash from real troff input.
*
* I believe Plan 9 troff allows non-ASCII characters in the names
* of macros, so this test might possibly fail on such a file.
*/
if (*ubuf == '.') {
my_unichar *tp = ubuf + 1;
while (ISSPC(*tp))
++tp; /* skip leading whitespace */
if ((tp[0] == '\\' && tp[1] == '\"') ||
(isascii((unsigned char)tp[0]) &&
isalnum((unsigned char)tp[0]) &&
isascii((unsigned char)tp[1]) &&
isalnum((unsigned char)tp[1]) &&
ISSPC(tp[2]))) {
subtype_mime = "text/troff";
subtype = "troff or preprocessor input";
goto subtype_identified;
}
}
if ((*buf == 'c' || *buf == 'C') && ISSPC(buf[1])) {
subtype_mime = "text/fortran";
subtype = "fortran program";
goto subtype_identified;
}
/* look for tokens from names.h - this is expensive! */
i = 0;
while (i < ulen) {
size_t end;
/*
* skip past any leading space
*/
while (i < ulen && ISSPC(ubuf[i]))
i++;
if (i >= ulen)
break;
/*
* find the next whitespace
*/
for (end = i + 1; end < nbytes; end++)
if (ISSPC(ubuf[end]))
break;
/*
* compare the word thus isolated against the token list
*/
for (p = names; p < names + NNAMES; p++) {
if (ascmatch((const unsigned char *)p->name, ubuf + i,
end - i)) {
subtype = types[p->type].human;
subtype_mime = types[p->type].mime;
goto subtype_identified;
}
}
i = end;
}
subtype_identified:
/*
* Now try to discover other details about the file.
*/
for (i = 0; i < ulen; i++) {
if (ubuf[i] == '\n') {
if (seen_cr)
n_crlf++;
else
n_lf++;
last_line_end = i;
} else if (seen_cr)
n_cr++;
seen_cr = (ubuf[i] == '\r');
if (seen_cr)
last_line_end = i;
if (ubuf[i] == 0x85) { /* X3.64/ECMA-43 "next line" character */
n_nel++;
last_line_end = i;
}
/* If this line is _longer_ than MAXLINELEN, remember it. */
if ((int)i > last_line_end + MAXLINELEN)
has_long_lines = 1;
if (ubuf[i] == '\033')
has_escapes = 1;
if (ubuf[i] == '\b')
has_backspace = 1;
}
rv = 1;
done:
if (nbuf)
free(nbuf);
if (ubuf)
free(ubuf);
if (rv) {
// If we have identified the subtype, return it, otherwise just
// text/plain.
if (subtype_mime)
mimeType->SetTo(subtype_mime);
else
mimeType->SetTo("text/plain");
}
return rv;
}
static int
ascmatch(const unsigned char *s, const my_unichar *us, size_t ulen)
{
size_t i;
for (i = 0; i < ulen; i++) {
if (s[i] != us[i])
return 0;
}
if (s[i])
return 0;
else
return 1;
}
/*
* This table reflects a particular philosophy about what constitutes
* "text," and there is room for disagreement about it.
*
* Version 3.31 of the file command considered a file to be ASCII if
* each of its characters was approved by either the isascii() or
* isalpha() function. On most systems, this would mean that any
* file consisting only of characters in the range 0x00 ... 0x7F
* would be called ASCII text, but many systems might reasonably
* consider some characters outside this range to be alphabetic,
* so the file command would call such characters ASCII. It might
* have been more accurate to call this "considered textual on the
* local system" than "ASCII."
*
* It considered a file to be "International language text" if each
* of its characters was either an ASCII printing character (according
* to the real ASCII standard, not the above test), a character in
* the range 0x80 ... 0xFF, or one of the following control characters:
* backspace, tab, line feed, vertical tab, form feed, carriage return,
* escape. No attempt was made to determine the language in which files
* of this type were written.
*
*
* The table below considers a file to be ASCII if all of its characters
* are either ASCII printing characters (again, according to the X3.4
* standard, not isascii()) or any of the following controls: bell,
* backspace, tab, line feed, form feed, carriage return, esc, nextline.
*
* I include bell because some programs (particularly shell scripts)
* use it literally, even though it is rare in normal text. I exclude
* vertical tab because it never seems to be used in real text. I also
* include, with hesitation, the X3.64/ECMA-43 control nextline (0x85),
* because that's what the dd EBCDIC->ASCII table maps the EBCDIC newline
* character to. It might be more appropriate to include it in the 8859
* set instead of the ASCII set, but it's got to be included in *something*
* we recognize or EBCDIC files aren't going to be considered textual.
* Some old Unix source files use SO/SI (^N/^O) to shift between Greek
* and Latin characters, so these should possibly be allowed. But they
* make a real mess on VT100-style displays if they're not paired properly,
* so we are probably better off not calling them text.
*
* A file is considered to be ISO-8859 text if its characters are all
* either ASCII, according to the above definition, or printing characters
* from the ISO-8859 8-bit extension, characters 0xA0 ... 0xFF.
*
* Finally, a file is considered to be international text from some other
* character code if its characters are all either ISO-8859 (according to
* the above definition) or characters in the range 0x80 ... 0x9F, which
* ISO-8859 considers to be control characters but the IBM PC and Macintosh
* consider to be printing characters.
*/
#define F 0 /* character never appears in text */
#define T 1 /* character appears in plain ASCII text */
#define I 2 /* character appears in ISO-8859 text */
#define X 3 /* character appears in non-ISO extended ASCII (Mac, IBM PC) */
static char text_chars[256] = {
/* BEL BS HT LF FF CR */
F, F, F, F, F, F, F, T, T, T, T, F, T, T, F, F, /* 0x0X */
/* ESC */
F, F, F, F, F, F, F, F, F, F, F, T, F, F, F, F, /* 0x1X */
T, T, T, T, T, T, T, T, T, T, T, T, T, T, T, T, /* 0x2X */
T, T, T, T, T, T, T, T, T, T, T, T, T, T, T, T, /* 0x3X */
T, T, T, T, T, T, T, T, T, T, T, T, T, T, T, T, /* 0x4X */
T, T, T, T, T, T, T, T, T, T, T, T, T, T, T, T, /* 0x5X */
T, T, T, T, T, T, T, T, T, T, T, T, T, T, T, T, /* 0x6X */
T, T, T, T, T, T, T, T, T, T, T, T, T, T, T, F, /* 0x7X */
/* NEL */
X, X, X, X, X, T, X, X, X, X, X, X, X, X, X, X, /* 0x8X */
X, X, X, X, X, X, X, X, X, X, X, X, X, X, X, X, /* 0x9X */
I, I, I, I, I, I, I, I, I, I, I, I, I, I, I, I, /* 0xaX */
I, I, I, I, I, I, I, I, I, I, I, I, I, I, I, I, /* 0xbX */
I, I, I, I, I, I, I, I, I, I, I, I, I, I, I, I, /* 0xcX */
I, I, I, I, I, I, I, I, I, I, I, I, I, I, I, I, /* 0xdX */
I, I, I, I, I, I, I, I, I, I, I, I, I, I, I, I, /* 0xeX */
I, I, I, I, I, I, I, I, I, I, I, I, I, I, I, I /* 0xfX */
};
static int
looks_ascii(const unsigned char *buf, size_t nbytes, my_unichar *ubuf,
size_t *ulen)
{
int i;
*ulen = 0;
for (i = 0; i < (int)nbytes; i++) {
int t = text_chars[buf[i]];
if (t != T)
return 0;
ubuf[(*ulen)++] = buf[i];
}
return 1;
}
static int
looks_latin1(const unsigned char *buf, size_t nbytes, my_unichar *ubuf, size_t *ulen)
{
int i;
*ulen = 0;
for (i = 0; i < (int)nbytes; i++) {
int t = text_chars[buf[i]];
if (t != T && t != I)
return 0;
ubuf[(*ulen)++] = buf[i];
}
return 1;
}
static int
looks_extended(const unsigned char *buf, size_t nbytes, my_unichar *ubuf,
size_t *ulen)
{
int i;
*ulen = 0;
for (i = 0; i < (int)nbytes; i++) {
int t = text_chars[buf[i]];
if (t != T && t != I && t != X)
return 0;
ubuf[(*ulen)++] = buf[i];
}
return 1;
}
static int
looks_utf8(const unsigned char *buf, size_t nbytes, my_unichar *ubuf, size_t *ulen)
{
int i, n;
my_unichar c;
int gotone = 0;
*ulen = 0;
for (i = 0; i < (int)nbytes; i++) {
if ((buf[i] & 0x80) == 0) { /* 0xxxxxxx is plain ASCII */
/*
* Even if the whole file is valid UTF-8 sequences,
* still reject it if it uses weird control characters.
*/
if (text_chars[buf[i]] != T)
return 0;
ubuf[(*ulen)++] = buf[i];
} else if ((buf[i] & 0x40) == 0) { /* 10xxxxxx never 1st byte */
return 0;
} else { /* 11xxxxxx begins UTF-8 */
int following;
if ((buf[i] & 0x20) == 0) { /* 110xxxxx */
c = buf[i] & 0x1f;
following = 1;
} else if ((buf[i] & 0x10) == 0) { /* 1110xxxx */
c = buf[i] & 0x0f;
following = 2;
} else if ((buf[i] & 0x08) == 0) { /* 11110xxx */
c = buf[i] & 0x07;
following = 3;
} else if ((buf[i] & 0x04) == 0) { /* 111110xx */
c = buf[i] & 0x03;
following = 4;
} else if ((buf[i] & 0x02) == 0) { /* 1111110x */
c = buf[i] & 0x01;
following = 5;
} else
return 0;
for (n = 0; n < following; n++) {
i++;
if (i >= (int)nbytes)
goto done;
if ((buf[i] & 0x80) == 0 || (buf[i] & 0x40))
return 0;
c = (c << 6) + (buf[i] & 0x3f);
}
ubuf[(*ulen)++] = c;
gotone = 1;
}
}
done:
return gotone; /* don't claim it's UTF-8 if it's all 7-bit */
}
static int
looks_unicode(const unsigned char *buf, size_t nbytes, my_unichar *ubuf,
size_t *ulen)
{
int bigend;
int i;
if (nbytes < 2)
return 0;
if (buf[0] == 0xff && buf[1] == 0xfe)
bigend = 0;
else if (buf[0] == 0xfe && buf[1] == 0xff)
bigend = 1;
else
return 0;
*ulen = 0;
for (i = 2; i + 1 < (int)nbytes; i += 2) {
/* XXX fix to properly handle chars > 65536 */
if (bigend)
ubuf[(*ulen)++] = buf[i + 1] + 256 * buf[i];
else
ubuf[(*ulen)++] = buf[i] + 256 * buf[i + 1];
if (ubuf[*ulen - 1] == 0xfffe)
return 0;
if (ubuf[*ulen - 1] < 128 &&
text_chars[(size_t)ubuf[*ulen - 1]] != T)
return 0;
}
return 1 + bigend;
}
#undef F
#undef T
#undef I
#undef X
/*
* This table maps each EBCDIC character to an (8-bit extended) ASCII
* character, as specified in the rationale for the dd(1) command in
* draft 11.2 (September, 1991) of the POSIX P1003.2 standard.
*
* Unfortunately it does not seem to correspond exactly to any of the
* five variants of EBCDIC documented in IBM's _Enterprise Systems
* Architecture/390: Principles of Operation_, SA22-7201-06, Seventh
* Edition, July, 1999, pp. I-1 - I-4.
*
* Fortunately, though, all versions of EBCDIC, including this one, agree
* on most of the printing characters that also appear in (7-bit) ASCII.
* Of these, only '|', '!', '~', '^', '[', and ']' are in question at all.
*
* Fortunately too, there is general agreement that codes 0x00 through
* 0x3F represent control characters, 0x41 a nonbreaking space, and the
* remainder printing characters.
*
* This is sufficient to allow us to identify EBCDIC text and to distinguish
* between old-style and internationalized examples of text.
*/
static unsigned char ebcdic_to_ascii[] = {
0, 1, 2, 3, 156, 9, 134, 127, 151, 141, 142, 11, 12, 13, 14, 15,
16, 17, 18, 19, 157, 133, 8, 135, 24, 25, 146, 143, 28, 29, 30, 31,
128, 129, 130, 131, 132, 10, 23, 27, 136, 137, 138, 139, 140, 5, 6, 7,
144, 145, 22, 147, 148, 149, 150, 4, 152, 153, 154, 155, 20, 21, 158, 26,
' ', 160, 161, 162, 163, 164, 165, 166, 167, 168, 213, '.', '<', '(', '+', '|',
'&', 169, 170, 171, 172, 173, 174, 175, 176, 177, '!', '$', '*', ')', ';', '~',
'-', '/', 178, 179, 180, 181, 182, 183, 184, 185, 203, ',', '%', '_', '>', '?',
186, 187, 188, 189, 190, 191, 192, 193, 194, '`', ':', '#', '@', '\'','=', '"',
195, 'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', 196, 197, 198, 199, 200, 201,
202, 'j', 'k', 'l', 'm', 'n', 'o', 'p', 'q', 'r', '^', 204, 205, 206, 207, 208,
209, 229, 's', 't', 'u', 'v', 'w', 'x', 'y', 'z', 210, 211, 212, '[', 214, 215,
216, 217, 218, 219, 220, 221, 222, 223, 224, 225, 226, 227, 228, ']', 230, 231,
'{', 'A', 'B', 'C', 'D', 'E', 'F', 'G', 'H', 'I', 232, 233, 234, 235, 236, 237,
'}', 'J', 'K', 'L', 'M', 'N', 'O', 'P', 'Q', 'R', 238, 239, 240, 241, 242, 243,
'\\',159, 'S', 'T', 'U', 'V', 'W', 'X', 'Y', 'Z', 244, 245, 246, 247, 248, 249,
'0', '1', '2', '3', '4', '5', '6', '7', '8', '9', 250, 251, 252, 253, 254, 255
};
#ifdef notdef
/*
* The following EBCDIC-to-ASCII table may relate more closely to reality,
* or at least to modern reality. It comes from
*
* http://ftp.s390.ibm.com/products/oe/bpxqp9.html
*
* and maps the characters of EBCDIC code page 1047 (the code used for
* Unix-derived software on IBM's 390 systems) to the corresponding
* characters from ISO 8859-1.
*
* If this table is used instead of the above one, some of the special
* cases for the NEL character can be taken out of the code.
*/
static unsigned char ebcdic_1047_to_8859[] = {
0x00,0x01,0x02,0x03,0x9C,0x09,0x86,0x7F,0x97,0x8D,0x8E,0x0B,0x0C,0x0D,0x0E,0x0F,
0x10,0x11,0x12,0x13,0x9D,0x0A,0x08,0x87,0x18,0x19,0x92,0x8F,0x1C,0x1D,0x1E,0x1F,
0x80,0x81,0x82,0x83,0x84,0x85,0x17,0x1B,0x88,0x89,0x8A,0x8B,0x8C,0x05,0x06,0x07,
0x90,0x91,0x16,0x93,0x94,0x95,0x96,0x04,0x98,0x99,0x9A,0x9B,0x14,0x15,0x9E,0x1A,
0x20,0xA0,0xE2,0xE4,0xE0,0xE1,0xE3,0xE5,0xE7,0xF1,0xA2,0x2E,0x3C,0x28,0x2B,0x7C,
0x26,0xE9,0xEA,0xEB,0xE8,0xED,0xEE,0xEF,0xEC,0xDF,0x21,0x24,0x2A,0x29,0x3B,0x5E,
0x2D,0x2F,0xC2,0xC4,0xC0,0xC1,0xC3,0xC5,0xC7,0xD1,0xA6,0x2C,0x25,0x5F,0x3E,0x3F,
0xF8,0xC9,0xCA,0xCB,0xC8,0xCD,0xCE,0xCF,0xCC,0x60,0x3A,0x23,0x40,0x27,0x3D,0x22,
0xD8,0x61,0x62,0x63,0x64,0x65,0x66,0x67,0x68,0x69,0xAB,0xBB,0xF0,0xFD,0xFE,0xB1,
0xB0,0x6A,0x6B,0x6C,0x6D,0x6E,0x6F,0x70,0x71,0x72,0xAA,0xBA,0xE6,0xB8,0xC6,0xA4,
0xB5,0x7E,0x73,0x74,0x75,0x76,0x77,0x78,0x79,0x7A,0xA1,0xBF,0xD0,0x5B,0xDE,0xAE,
0xAC,0xA3,0xA5,0xB7,0xA9,0xA7,0xB6,0xBC,0xBD,0xBE,0xDD,0xA8,0xAF,0x5D,0xB4,0xD7,
0x7B,0x41,0x42,0x43,0x44,0x45,0x46,0x47,0x48,0x49,0xAD,0xF4,0xF6,0xF2,0xF3,0xF5,
0x7D,0x4A,0x4B,0x4C,0x4D,0x4E,0x4F,0x50,0x51,0x52,0xB9,0xFB,0xFC,0xF9,0xFA,0xFF,
0x5C,0xF7,0x53,0x54,0x55,0x56,0x57,0x58,0x59,0x5A,0xB2,0xD4,0xD6,0xD2,0xD3,0xD5,
0x30,0x31,0x32,0x33,0x34,0x35,0x36,0x37,0x38,0x39,0xB3,0xDB,0xDC,0xD9,0xDA,0x9F
};
#endif
/*
* Copy buf[0 ... nbytes-1] into out[], translating EBCDIC to ASCII.
*/
static void
from_ebcdic(const unsigned char *buf, size_t nbytes, unsigned char *out)
{
int i;
for (i = 0; i < (int)nbytes; i++) {
out[i] = ebcdic_to_ascii[buf[i]];
}
}
+24
View File
@@ -0,0 +1,24 @@
/*
* Copyright 2006, Ingo Weinhold <bonefish@cs.tu-berlin.de>.
* All rights reserved. Distributed under the terms of the MIT License.
*/
#ifndef TEXT_SNIFFER_ADDON_H
#define TEXT_SNIFFER_ADDON_H
#include <MimeSnifferAddon.h>
class TextSnifferAddon : public BMimeSnifferAddon {
public:
TextSnifferAddon();
virtual ~TextSnifferAddon();
virtual size_t MinimalBufferSize();
virtual float GuessMimeType(const char* fileName,
BMimeType* type);
virtual float GuessMimeType(BFile* file,
const void* buffer, int32 length,
BMimeType* type);
};
#endif // TEXT_SNIFFER_ADDON_H
+186
View File
@@ -0,0 +1,186 @@
/*
* Copyright (c) Ian F. Darwin 1986-1995.
* Software written by Ian F. Darwin and others;
* maintained 1995-present by Christos Zoulas and others.
*
* Redistribution and use in source and binary forms, with or without
* modification, are permitted provided that the following conditions
* are met:
* 1. Redistributions of source code must retain the above copyright
* notice immediately at the beginning of the file, without modification,
* this list of conditions, and the following disclaimer.
* 2. Redistributions in binary form must reproduce the above copyright
* notice, this list of conditions and the following disclaimer in the
* documentation and/or other materials provided with the distribution.
*
* THIS SOFTWARE IS PROVIDED BY THE AUTHOR AND CONTRIBUTORS ``AS IS'' AND
* ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
* ARE DISCLAIMED. IN NO EVENT SHALL THE AUTHOR OR CONTRIBUTORS BE LIABLE FOR
* ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
* OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
* HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
* LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
* OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
* SUCH DAMAGE.
*/
/*
* Names.h - names and types used by ascmagic in file(1).
* These tokens are here because they can appear anywhere in
* the first HOWMANY bytes, while tokens in MAGIC must
* appear at fixed offsets into the file. Don't make HOWMANY
* too high unless you have a very fast CPU.
*
* $Id: names.h,v 1.25 2004/09/11 19:15:57 christos Exp $
*/
/*
modified by Chris Lowth - 9 April 2000
to add mime type strings to the types table.
*/
/* these types are used to index the table 'types': keep em in sync! */
#define L_C 0 /* first and foremost on UNIX */
#define L_CC 1 /* Bjarne's postincrement */
#define L_FORT 2 /* the oldest one */
#define L_MAKE 3 /* Makefiles */
#define L_PLI 4 /* PL/1 */
#define L_MACH 5 /* some kinda assembler */
#define L_ENG 6 /* English */
#define L_PAS 7 /* Pascal */
#define L_MAIL 8 /* Electronic mail */
#define L_NEWS 9 /* Usenet Netnews */
#define L_JAVA 10 /* Java code */
#define L_HTML 11 /* HTML */
#define L_BCPL 12 /* BCPL */
#define L_M4 13 /* M4 */
#define L_PO 14 /* PO */
static const struct {
const char *human;
const char *mime;
} types[] = {
{ "C program", "text/x-c", },
{ "C++ program", "text/x-c++" },
{ "FORTRAN program", "text/x-fortran" },
{ "make commands", "text/x-makefile" },
{ "PL/1 program", "text/x-pl1" },
{ "assembler program", "text/x-asm" },
{ "English", "text/plain" },
{ "Pascal program", "text/x-pascal" },
{ "mail", "text/x-mail" },
{ "news", "text/x-news" },
{ "Java program", "text/x-java" },
{ "HTML document", "text/html", },
{ "BCPL program", "text/x-bcpl" },
{ "M4 macro language pre-processor", "text/x-m4" },
{ "PO (gettext message catalogue)", "text/x-po" },
{ "cannot happen error on names.h/types", "error/x-error" },
{ 0, 0}
};
/*
* XXX - how should we distinguish Java from C++?
* The trick used in a Debian snapshot, of having "extends" or "implements"
* as tags for Java, doesn't work very well, given that those keywords
* are often preceded by "class", which flags it as C++.
*
* Perhaps we need to be able to say
*
* If "class" then
*
* if "extends" or "implements" then
* Java
* else
* C++
* endif
*
* Or should we use other keywords, such as "package" or "import"?
* Unfortunately, Ada95 uses "package", and Modula-3 uses "import",
* although I infer from the language spec at
*
* http://www.research.digital.com/SRC/m3defn/html/m3.html
*
* that Modula-3 uses "IMPORT" rather than "import", i.e. it must be
* in all caps.
*
* So, for now, we go with "import". We must put it before the C++
* stuff, so that we don't misidentify Java as C++. Not using "package"
* means we won't identify stuff that defines a package but imports
* nothing; hopefully, very little Java code imports nothing (one of the
* reasons for doing OO programming is to import as much as possible
* and write only what you need to, right?).
*
* Unfortunately, "import" may cause us to misidentify English text
* as Java, as it comes after "the" and "The". Perhaps we need a fancier
* heuristic to identify Java?
*/
static struct names {
const char *name;
short type;
} names[] = {
/* These must be sorted by eye for optimal hit rate */
/* Add to this list only after substantial meditation */
{"msgid", L_PO},
{"dnl", L_M4},
{"import", L_JAVA},
{"\"libhdr\"", L_BCPL},
{"\"LIBHDR\"", L_BCPL},
{"//", L_CC},
{"template", L_CC},
{"virtual", L_CC},
{"class", L_CC},
{"public:", L_CC},
{"private:", L_CC},
{"/*", L_C}, /* must precede "The", "the", etc. */
{"#include", L_C},
{"char", L_C},
{"The", L_ENG},
{"the", L_ENG},
{"double", L_C},
{"extern", L_C},
{"float", L_C},
{"struct", L_C},
{"union", L_C},
{"CFLAGS", L_MAKE},
{"LDFLAGS", L_MAKE},
{"all:", L_MAKE},
{".PRECIOUS", L_MAKE},
/* Too many files of text have these words in them. Find another way
* to recognize Fortrash.
*/
#ifdef NOTDEF
{"subroutine", L_FORT},
{"function", L_FORT},
{"block", L_FORT},
{"common", L_FORT},
{"dimension", L_FORT},
{"integer", L_FORT},
{"data", L_FORT},
#endif /*NOTDEF*/
{".ascii", L_MACH},
{".asciiz", L_MACH},
{".byte", L_MACH},
{".even", L_MACH},
{".globl", L_MACH},
{".text", L_MACH},
{"clr", L_MACH},
{"(input,", L_PAS},
{"dcl", L_PLI},
{"Received:", L_MAIL},
{">From", L_MAIL},
{"Return-Path:",L_MAIL},
{"Cc:", L_MAIL},
{"Newsgroups:", L_NEWS},
{"Path:", L_NEWS},
{"Organization:",L_NEWS},
{"href=", L_HTML},
{"HREF=", L_HTML},
{"<body", L_HTML},
{"<BODY", L_HTML},
{"<html", L_HTML},
{"<HTML", L_HTML},
{NULL, 0}
};
#define NNAMES ((sizeof(names)/sizeof(struct names)) - 1)