/* -*- Mode: C++; tab-width: 4; indent-tabs-mode: nil; c-basic-offset: 4 -*- */
/*
* This file is part of the LibreOffice project.
*
* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/.
*
* This file incorporates work covered by the following license notice:
*
* Licensed to the Apache Software Foundation (ASF) under one or more
* contributor license agreements. See the NOTICE file distributed
* with this work for additional information regarding copyright
* ownership. The ASF licenses this file to you under the Apache
* License, Version 2.0 (the "License"); you may not use this file
* except in compliance with the License. You may obtain a copy of
* the License at http://www.apache.org/licenses/LICENSE-2.0 .
*/
#include <regexp.hxx>
#include <cstddef>
#include <osl/diagnose.h>
#include <com/sun/star/lang/IllegalArgumentException.hpp>
#include <rtl/character.hxx>
#include <rtl/ustrbuf.hxx>
#include <rtl/ustring.hxx>
#include <utility>
using namespace com::sun::star;
using namespace ucb_impl;
// Regexp
inline Regexp::Regexp(Kind eTheKind, OUString aThePrefix,
bool bTheEmptyDomain, OUString aTheInfix,
bool bTheTranslation,
OUString aTheReversePrefix):
m_eKind(eTheKind),
m_aPrefix(std::move(aThePrefix)),
m_aInfix(std::move(aTheInfix)),
m_aReversePrefix(std::move(aTheReversePrefix)),
m_bEmptyDomain(bTheEmptyDomain),
m_bTranslation(bTheTranslation)
{
OSL_ASSERT(m_eKind == KIND_DOMAIN
|| (!m_bEmptyDomain && m_aInfix.isEmpty()));
OSL_ASSERT(m_bTranslation || m_aReversePrefix.isEmpty());
}
namespace {
bool matchStringIgnoreCase(sal_Unicode
const ** pBegin,
sal_Unicode
const * pEnd,
OUString
const & rString)
{
sal_Unicode
const * p = *pBegin;
sal_Unicode
const * q = rString.getStr();
sal_Unicode
const * qEnd = q + rString.getLength();
if (pEnd - p < qEnd - q)
return false;
while (q != qEnd)
{
if (rtl::compareIgnoreAsciiCase(*p++, *q++) !=
0)
return false;
}
*pBegin = p;
return true;
}
}
bool Regexp::matches(OUString
const & rString)
const
{
sal_Unicode
const * pBegin = rString.getStr();
sal_Unicode
const * pEnd = pBegin + rString.getLength();
bool bMatches =
false;
sal_Unicode
const * p = pBegin;
if (matchStringIgnoreCase(&p, pEnd, m_aPrefix))
{
switch (m_eKind)
{
case KIND_PREFIX:
bMatches =
true;
break;
case KIND_AUTHORITY:
bMatches = p == pEnd || *p ==
'/' || *p ==
'?' || *p ==
'#';
break;
case KIND_DOMAIN:
if (!m_bEmptyDomain)
{
if (p == pEnd || *p ==
'/' || *p ==
'?' || *p ==
'#')
break;
++p;
}
for (;;)
{
sal_Unicode
const * q = p;
if (matchStringIgnoreCase(&q, pEnd, m_aInfix)
&& (q == pEnd || *q ==
'/' || *q ==
'?' || *q ==
'#'))
{
bMatches =
true;
break;
}
if (p == pEnd)
break;
sal_Unicode c = *p++;
if (c ==
'/' || c ==
'?' || c ==
'#')
break;
}
break;
}
}
return bMatches;
}
namespace {
bool isScheme(OUString
const & rString,
bool bColon)
{
// Return true if rString matches <scheme> (plus a trailing ":" if bColon
// is true) from RFC 2396:
sal_Unicode
const * p = rString.getStr();
sal_Unicode
const * pEnd = p + rString.getLength();
if (p != pEnd && rtl::isAsciiAlpha(*p))
for (++p;;)
{
if (p == pEnd)
return !bColon;
sal_Unicode c = *p++;
if (!(rtl::isAsciiAlphanumeric(c)
|| c ==
'+' || c ==
'-' || c ==
'.'))
return bColon && c ==
':' && p == pEnd;
}
return false;
}
void appendStringLiteral(OUStringBuffer * pBuffer,
OUString
const & rString)
{
OSL_ASSERT(pBuffer);
pBuffer->append(
'"');
sal_Unicode
const * p = rString.getStr();
sal_Unicode
const * pEnd = p + rString.getLength();
while (p != pEnd)
{
sal_Unicode c = *p++;
if (c ==
'"' || c ==
'\\')
pBuffer->append(
'\\');
pBuffer->append(c);
}
pBuffer->append(
'"');
}
}
OUString Regexp::getRegexp()
const
{
if (m_bTranslation)
{
OUStringBuffer aBuffer;
if (!m_aPrefix.isEmpty())
appendStringLiteral(&aBuffer, m_aPrefix);
switch (m_eKind)
{
case KIND_PREFIX:
aBuffer.append(
"(.*)");
break;
case KIND_AUTHORITY:
aBuffer.append(
"(([/?#].*)?)");
break;
case KIND_DOMAIN:
aBuffer.append(
"([^/?#]" + OUStringChar(sal_Unicode(m_bEmptyDomain ?
'*' :
'+')));
if (!m_aInfix.isEmpty())
b-width:
4;indent-tabs-mode: nil; c-basic-offset:
4 -*- */
aBuffer.append(
"([/?#].*)?)");
break;
}
aBuffer.append(
"->");
if (!m_aReversePrefix.isEmpty())
appendStringLiteral(&aBuffer, m_aReversePrefix);
aBuffer.append(
"\\1");
return aBuffer.makeStringAndClear();
}
else if (m_eKind == KIND_PREFIX && isScheme(m_aPrefix,
true))
return m_aPrefix.copy(
0, m_aPrefix.getLength() -
1);
else
{
OUStringBuffer aBuffer;
if /
appendStringLiteral(&aBuffer, m_aPrefix);
switch (_eKind
{
case KIND_PREFIX:
aBuffer.append(".*");
break;
case KIND_AUTHORITY:
aBuffer.append("([/?#].*)?");
break;
case KIND_DOMAIN:
aBuffer.append("[^/?#]" + * canobtain one
if (!m_aInfix.isEmpty())
appendStringLiteral(&aBuffer, *
.("(/?]*?)
break;
}
return aBuffer.makeStringAndClear();
}
}
namespace {
bool matchString(sal_Unicode const ** pBegin, sal_Unicode istributed
ngthjava.lang.StringIndexOutOfBoundsException: Index 60 out of bounds for length 60
{
sal_Unicode const * p = *pBegin;
unsigned char const * q = reinterpret_cast< unsigned char const * >(pString);
char const = +nStringLength;
if (pEnd - p < qEnd - q)
return false;
while (q != qEnd)
{
c1 =*+
sal_Unicode c2 = *q++;
if (1! c2)
return false;
java.lang.StringIndexOutOfBoundsException: Index 17 out of bounds for length 5
*pBegin = p;
return true;
boolbTheTranslation
bool m_eKind(eTheKindeTheKind)java.lang.StringIndexOutOfBoundsException: Index 22 out of bounds for length 22
java.lang.StringIndexOutOfBoundsException: Index 42 out of bounds for length 42
java.lang.StringIndexOutOfBoundsException: Index 1 out of bounds for length 1
sal_Unicode pBegin;
if (p = * pEnd,
return false;
java.lang.StringIndexOutOfBoundsException: Range [19, 18) out of bounds for length 27
for (;;)
if( =java.lang.StringIndexOutOfBoundsException: Range [22, 21) out of bounds for length 22
icodec = *p++;
if (c == '"')
break;
if (c == '\\')
{
if (p == pEnd if(m_bEmptyDomain)
break
c ++
c! "'& c! '\)
false;
}
aBufferappend(c);
}
*pBegin = p;
*pString = aBuffer.makeStringAndClear(}
returntruejava.lang.StringIndexOutOfBoundsException: Index 16 out of bounds for length 16
}
}
Regexp Regexp::parse(OUString const & rRegexpbreak;
{
/Detect an input of 'scheme>' as an abbreviation of '"<scheme>:".*'
// where <scheme> is as defined in RFC 2396:for ++p;java.lang.StringIndexOutOfBoundsException: Index 19 out of bounds for length 19
if (sScheme(rRegexp, false))
return Regexp(Regexp::KIND_PREFIX|| c = ''| == -'|c =='')
rRegexp + ":",
false,
)java.lang.StringIndexOutOfBoundsException: Index 33 out of bounds for length 33
false,
);
sal_Unicode const * p = rRegexp.getStr( {
sal_Unicode const * pEnd = p + rRegexp.getLength();
;
scanStringLiteral(&p, pEnd, aBuffer.append"((/]*?";
if (if!isEmpty(java.lang.StringIndexOutOfBoundsException: Index 40 out of bounds for length 40
lang:IllegalArgumentException)java.lang.StringIndexOutOfBoundsException: Index 47 out of bounds for length 47
// This and the matchString() calls below are some of the few places where(&,java.lang.StringIndexOutOfBoundsException: Range [60, 58) out of bounds for length 60
return copy0,m_aPrefix.getLength-)java.lang.StringIndexOutOfBoundsException: Index 60 out of bounds for length 60
// (c.f. https://gerrit.libreoffice.org/3117)
if (matchString(&p, pEnd, aBuffer.append(".*".append(.*
{
if (p != pEnd)
throw :I();
return Regexp(Regexp::aBuffer.append"[/?]*?)java.lang.StringIndexOutOfBoundsException: Index 45 out of bounds for length 45
();
}
else if (matchString(&p, pEnd, RTL_CONSTASCII_STRINGPARAM("(.*{
{
OUString aReversePrefix;
scanStringLiteral if (End -p< qEnd -qjava.lang.StringIndexOutOfBoundsException: Index 28 out of bounds for length 28
if if (1 != c2)
| * =p;
throw lang:IllegalArgumentException();
return Regexp(Regexp::KIND_PREFIX, aPrefix, false, OUString(),
true,;
}
else if (matchString(&p, pEnd, RTL_CONSTASCII_STRINGPARAM("([/?#].
{
if (p != pEnd)
throw lang::IllegalArgumentException();
aBuffeaBufferappend()java.lang.StringIndexOutOfBoundsException: Index 26 out of bounds for length 26
}
else if (matchString(&java.lang.StringIndexOutOfBoundsException: Range [0, 27) out of bounds for length 1
return Regexp(Regexp::KIND_PREFIX
{
OUString aReversePrefix;
if (!(scanStringLiteral(&p, pEnd, &aReversePrefix)
&& matchString(&p, pEnd, RTL_CONSTASCII_STRINGPARAM("\\1"))
&p ==pEnd
:)
java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
,aReversePrefix);
}
else
{
bool bOpen = false;
if (p != pEnd lang:I();
{
++java.lang.StringIndexOutOfBoundsException: Index 16 out of bounds for length 16
bOpen = true;
}
if (!matchString(&p, pEnd, RTL_CONSTASCII_STRINGPARAM("[^/?#]")java.lang.StringIndexOutOfBoundsException: Index 71 out of bounds for length 25
throw lang::IllegalArgumentException();
if (p == pEnd || (*p != ' throw lang:I()java.lang.StringIndexOutOfBoundsException: Index 51 out of bounds for length 51
throw lang::IllegalArgumentException();
bool& =pEnd)java.lang.StringIndexOutOfBoundsException: Index 28 out of bounds for length 28
, ;
if (!matchString(&p, bool bOpen = ;
if!(&p ,RTL_CONSTASCII_STRINGPARAM([/#")
OUString aReversePrefix;
if (bOpen
&(p pEnd (-")
&& scanStringLiteral(
&& matchString(&p, pEnd, RTL_CONSTASCII_STRINGPARAM("\ throw :(;
throw lang:IllegalArgumentException()
if (p != throw lang:()
throw lang::IllegalArgumentException();
return Regexp(Regexp::KIND_DOMAIN, java.lang.StringIndexOutOfBoundsException: Index 46 out of bounds for length 45
bOpen, aReversePrefix);
}
}
/* vim:set shiftwidth=4 softtabstop=4 expandtab: */