Add poc files

This commit is contained in:
Joey Yakimowich-Payne 2018-05-26 10:55:38 +09:00
commit fa600b98f7
220 changed files with 45679 additions and 0 deletions

202
lang_cpp/parsing/.depend Normal file
View file

@ -0,0 +1,202 @@
ast_cpp.cmo : ../../h_program-lang/scope_code.cmi \
../../h_program-lang/parse_info.cmi ../../commons/common.cmi
ast_cpp.cmx : ../../h_program-lang/scope_code.cmx \
../../h_program-lang/parse_info.cmx ../../commons/common.cmx
flag_parsing_cpp.cmo : ../../globals/config_pfff.cmo
flag_parsing_cpp.cmx : ../../globals/config_pfff.cmx
lexer_cpp.cmo : parser_cpp.cmi ../../h_program-lang/parse_info.cmi \
flag_parsing_cpp.cmo ../../commons/common2.cmi ../../commons/common.cmi \
ast_cpp.cmo
lexer_cpp.cmx : parser_cpp.cmx ../../h_program-lang/parse_info.cmx \
flag_parsing_cpp.cmx ../../commons/common2.cmx ../../commons/common.cmx \
ast_cpp.cmx
lib_parsing_cpp.cmo : visitor_cpp.cmi ../../commons/file_type.cmi \
../../commons/common.cmi lib_parsing_cpp.cmi
lib_parsing_cpp.cmx : visitor_cpp.cmx ../../commons/file_type.cmx \
../../commons/common.cmx lib_parsing_cpp.cmi
lib_parsing_cpp.cmi : ../../h_program-lang/parse_info.cmi \
../../commons/common.cmi ast_cpp.cmo
meta_ast_cpp.cmo : ../../h_program-lang/scope_code.cmi \
../../h_program-lang/parse_info.cmi ../../commons/ocaml.cmi \
../../h_program-lang/meta_ast_generic.cmi ../../commons/common.cmi \
ast_cpp.cmo meta_ast_cpp.cmi
meta_ast_cpp.cmx : ../../h_program-lang/scope_code.cmx \
../../h_program-lang/parse_info.cmx ../../commons/ocaml.cmx \
../../h_program-lang/meta_ast_generic.cmx ../../commons/common.cmx \
ast_cpp.cmx meta_ast_cpp.cmi
meta_ast_cpp.cmi : ../../commons/ocaml.cmi \
../../h_program-lang/meta_ast_generic.cmi ast_cpp.cmo
parse_cpp.cmo : token_views_cpp.cmi token_helpers_cpp.cmi token_cpp.cmi \
pp_token.cmi parsing_recovery_cpp.cmi parsing_hacks_lib.cmi \
parsing_hacks_define.cmi parsing_hacks_cpp.cmi parsing_hacks.cmi \
parser_cpp_mly_helper.cmo parser_cpp.cmi \
../../h_program-lang/parse_info.cmi lexer_cpp.cmo flag_parsing_cpp.cmo \
../../commons/file_type.cmi ../../commons/common2.cmi \
../../commons/common.cmi ../../h_program-lang/ast_fuzzy.cmi ast_cpp.cmo \
parse_cpp.cmi
parse_cpp.cmx : token_views_cpp.cmx token_helpers_cpp.cmx token_cpp.cmx \
pp_token.cmx parsing_recovery_cpp.cmx parsing_hacks_lib.cmx \
parsing_hacks_define.cmx parsing_hacks_cpp.cmx parsing_hacks.cmx \
parser_cpp_mly_helper.cmx parser_cpp.cmx \
../../h_program-lang/parse_info.cmx lexer_cpp.cmx flag_parsing_cpp.cmx \
../../commons/file_type.cmx ../../commons/common2.cmx \
../../commons/common.cmx ../../h_program-lang/ast_fuzzy.cmx ast_cpp.cmx \
parse_cpp.cmi
parse_cpp.cmi : pp_token.cmi parser_cpp.cmi \
../../h_program-lang/parse_info.cmi flag_parsing_cpp.cmo \
../../commons/common.cmi ../../h_program-lang/ast_fuzzy.cmi ast_cpp.cmo
parser_cpp.cmo : token_cpp.cmi parser_cpp_mly_helper.cmo \
../../h_program-lang/parse_info.cmi ../../commons/common.cmi ast_cpp.cmo \
parser_cpp.cmi
parser_cpp.cmx : token_cpp.cmx parser_cpp_mly_helper.cmx \
../../h_program-lang/parse_info.cmx ../../commons/common.cmx ast_cpp.cmx \
parser_cpp.cmi
parser_cpp.cmi : token_cpp.cmi ../../h_program-lang/parse_info.cmi \
ast_cpp.cmo
parser_cpp_mly_helper.cmo : lib_parsing_cpp.cmi flag_parsing_cpp.cmo \
../../commons/common2.cmi ../../commons/common.cmi ast_cpp.cmo
parser_cpp_mly_helper.cmx : lib_parsing_cpp.cmx flag_parsing_cpp.cmx \
../../commons/common2.cmx ../../commons/common.cmx ast_cpp.cmx
parsing_hacks.cmo : token_views_cpp.cmi token_views_context.cmi \
token_helpers_cpp.cmi pp_token.cmi parsing_hacks_typedef.cmi \
parsing_hacks_pp.cmi parsing_hacks_define.cmi parsing_hacks_cpp.cmi \
parser_cpp.cmi ../../h_program-lang/parse_info.cmi flag_parsing_cpp.cmo \
../../commons/common2.cmi ../../commons/common.cmi ast_cpp.cmo \
parsing_hacks.cmi
parsing_hacks.cmx : token_views_cpp.cmx token_views_context.cmx \
token_helpers_cpp.cmx pp_token.cmx parsing_hacks_typedef.cmx \
parsing_hacks_pp.cmx parsing_hacks_define.cmx parsing_hacks_cpp.cmx \
parser_cpp.cmx ../../h_program-lang/parse_info.cmx flag_parsing_cpp.cmx \
../../commons/common2.cmx ../../commons/common.cmx ast_cpp.cmx \
parsing_hacks.cmi
parsing_hacks.cmi : pp_token.cmi parser_cpp.cmi flag_parsing_cpp.cmo
parsing_hacks_cpp.cmo : token_views_cpp.cmi token_helpers_cpp.cmi \
token_cpp.cmi parsing_hacks_lib.cmi parser_cpp.cmi \
../../h_program-lang/parse_info.cmi flag_parsing_cpp.cmo \
../../commons/common.cmi ast_cpp.cmo parsing_hacks_cpp.cmi
parsing_hacks_cpp.cmx : token_views_cpp.cmx token_helpers_cpp.cmx \
token_cpp.cmx parsing_hacks_lib.cmx parser_cpp.cmx \
../../h_program-lang/parse_info.cmx flag_parsing_cpp.cmx \
../../commons/common.cmx ast_cpp.cmx parsing_hacks_cpp.cmi
parsing_hacks_cpp.cmi : token_views_cpp.cmi
parsing_hacks_define.cmo : token_helpers_cpp.cmi parsing_hacks_lib.cmi \
parser_cpp.cmi ../../h_program-lang/parse_info.cmi flag_parsing_cpp.cmo \
../../commons/common2.cmi ../../commons/common.cmi ast_cpp.cmo \
parsing_hacks_define.cmi
parsing_hacks_define.cmx : token_helpers_cpp.cmx parsing_hacks_lib.cmx \
parser_cpp.cmx ../../h_program-lang/parse_info.cmx flag_parsing_cpp.cmx \
../../commons/common2.cmx ../../commons/common.cmx ast_cpp.cmx \
parsing_hacks_define.cmi
parsing_hacks_define.cmi : parser_cpp.cmi
parsing_hacks_lib.cmo : token_views_cpp.cmi token_helpers_cpp.cmi \
token_cpp.cmi parser_cpp.cmi ../../h_program-lang/parse_info.cmi \
flag_parsing_cpp.cmo ../../commons/common2.cmi ../../commons/common.cmi \
ast_cpp.cmo parsing_hacks_lib.cmi
parsing_hacks_lib.cmx : token_views_cpp.cmx token_helpers_cpp.cmx \
token_cpp.cmx parser_cpp.cmx ../../h_program-lang/parse_info.cmx \
flag_parsing_cpp.cmx ../../commons/common2.cmx ../../commons/common.cmx \
ast_cpp.cmx parsing_hacks_lib.cmi
parsing_hacks_lib.cmi : token_views_cpp.cmi token_cpp.cmi parser_cpp.cmi
parsing_hacks_pp.cmo : token_views_cpp.cmi token_helpers_cpp.cmi \
token_cpp.cmi parsing_hacks_lib.cmi parser_cpp.cmi \
../../h_program-lang/parse_info.cmi flag_parsing_cpp.cmo \
../../commons/common2.cmi ../../commons/common.cmi ast_cpp.cmo \
parsing_hacks_pp.cmi
parsing_hacks_pp.cmx : token_views_cpp.cmx token_helpers_cpp.cmx \
token_cpp.cmx parsing_hacks_lib.cmx parser_cpp.cmx \
../../h_program-lang/parse_info.cmx flag_parsing_cpp.cmx \
../../commons/common2.cmx ../../commons/common.cmx ast_cpp.cmx \
parsing_hacks_pp.cmi
parsing_hacks_pp.cmi : token_views_cpp.cmi
parsing_hacks_typedef.cmo : token_views_cpp.cmi token_views_context.cmi \
token_helpers_cpp.cmi parsing_hacks_lib.cmi parser_cpp.cmi \
../../h_program-lang/parse_info.cmi ../../commons/common.cmi ast_cpp.cmo \
parsing_hacks_typedef.cmi
parsing_hacks_typedef.cmx : token_views_cpp.cmx token_views_context.cmx \
token_helpers_cpp.cmx parsing_hacks_lib.cmx parser_cpp.cmx \
../../h_program-lang/parse_info.cmx ../../commons/common.cmx ast_cpp.cmx \
parsing_hacks_typedef.cmi
parsing_hacks_typedef.cmi : token_views_cpp.cmi
parsing_recovery_cpp.cmo : token_helpers_cpp.cmi parser_cpp.cmi \
../../h_program-lang/parse_info.cmi flag_parsing_cpp.cmo \
../../commons/common2.cmi ../../commons/common.cmi \
parsing_recovery_cpp.cmi
parsing_recovery_cpp.cmx : token_helpers_cpp.cmx parser_cpp.cmx \
../../h_program-lang/parse_info.cmx flag_parsing_cpp.cmx \
../../commons/common2.cmx ../../commons/common.cmx \
parsing_recovery_cpp.cmi
parsing_recovery_cpp.cmi : parser_cpp.cmi
pp_token.cmo : token_views_cpp.cmi token_helpers_cpp.cmi token_cpp.cmi \
parsing_hacks_lib.cmi parser_cpp.cmi flag_parsing_cpp.cmo \
../../commons/common2.cmi ../../commons/common.cmi ast_cpp.cmo \
pp_token.cmi
pp_token.cmx : token_views_cpp.cmx token_helpers_cpp.cmx token_cpp.cmx \
parsing_hacks_lib.cmx parser_cpp.cmx flag_parsing_cpp.cmx \
../../commons/common2.cmx ../../commons/common.cmx ast_cpp.cmx \
pp_token.cmi
pp_token.cmi : token_views_cpp.cmi parser_cpp.cmi ../../commons/common.cmi
test_dump_nim.cmo : ../../h_program-lang/parse_info.cmi parse_cpp.cmi \
flag_parsing_cpp.cmo ../../commons/common.cmi ast_cpp.cmo \
test_dump_nim.cmi
test_dump_nim.cmx : ../../h_program-lang/parse_info.cmx parse_cpp.cmx \
flag_parsing_cpp.cmx ../../commons/common.cmx ast_cpp.cmx \
test_dump_nim.cmi
test_dump_nim.cmi : ../../commons/common.cmi
test_parsing_cpp.cmo : token_views_cpp.cmi token_views_context.cmi \
token_helpers_cpp.cmi test_dump_nim.cmi \
../../h_program-lang/skip_code.cmi parsing_hacks_cpp.cmi parser_cpp.cmi \
../../h_program-lang/parse_info.cmi parse_cpp.cmi ../../commons/ocaml.cmi \
../../h_program-lang/meta_ast_generic.cmi meta_ast_cpp.cmi \
lib_parsing_cpp.cmi flag_parsing_cpp.cmo ../../commons_core/console.cmi \
../../commons/common.cmi ../../h_program-lang/ast_fuzzy.cmi ast_cpp.cmo \
test_parsing_cpp.cmi
test_parsing_cpp.cmx : token_views_cpp.cmx token_views_context.cmx \
token_helpers_cpp.cmx test_dump_nim.cmx \
../../h_program-lang/skip_code.cmx parsing_hacks_cpp.cmx parser_cpp.cmx \
../../h_program-lang/parse_info.cmx parse_cpp.cmx ../../commons/ocaml.cmx \
../../h_program-lang/meta_ast_generic.cmx meta_ast_cpp.cmx \
lib_parsing_cpp.cmx flag_parsing_cpp.cmx ../../commons_core/console.cmx \
../../commons/common.cmx ../../h_program-lang/ast_fuzzy.cmx ast_cpp.cmx \
test_parsing_cpp.cmi
test_parsing_cpp.cmi : ../../commons/common.cmi
token_cpp.cmo : token_cpp.cmi
token_cpp.cmx : token_cpp.cmi
token_cpp.cmi :
token_helpers_cpp.cmo : parser_cpp.cmi ../../h_program-lang/parse_info.cmi \
token_helpers_cpp.cmi
token_helpers_cpp.cmx : parser_cpp.cmx ../../h_program-lang/parse_info.cmx \
token_helpers_cpp.cmi
token_helpers_cpp.cmi : parser_cpp.cmi ../../h_program-lang/parse_info.cmi
token_views_context.cmo : token_views_cpp.cmi token_helpers_cpp.cmi \
parser_cpp.cmi ../../h_program-lang/parse_info.cmi \
../../commons/common2.cmi ../../commons/common.cmi \
token_views_context.cmi
token_views_context.cmx : token_views_cpp.cmx token_helpers_cpp.cmx \
parser_cpp.cmx ../../h_program-lang/parse_info.cmx \
../../commons/common2.cmx ../../commons/common.cmx \
token_views_context.cmi
token_views_context.cmi : token_views_cpp.cmi
token_views_cpp.cmo : token_helpers_cpp.cmi parser_cpp.cmi \
../../h_program-lang/parse_info.cmi ../../commons/ocaml.cmi \
flag_parsing_cpp.cmo ../../commons/common2.cmi ../../commons/common.cmi \
token_views_cpp.cmi
token_views_cpp.cmx : token_helpers_cpp.cmx parser_cpp.cmx \
../../h_program-lang/parse_info.cmx ../../commons/ocaml.cmx \
flag_parsing_cpp.cmx ../../commons/common2.cmx ../../commons/common.cmx \
token_views_cpp.cmi
token_views_cpp.cmi : parser_cpp.cmi ../../commons/ocaml.cmi
type_cpp.cmo : ast_cpp.cmo type_cpp.cmi
type_cpp.cmx : ast_cpp.cmx type_cpp.cmi
type_cpp.cmi : ast_cpp.cmo
unit_parsing_cpp.cmo : parse_cpp.cmi ../../commons/oUnit.cmi \
flag_parsing_cpp.cmo ../../globals/config_pfff.cmo \
../../commons/common2.cmi ../../commons/common.cmi ast_cpp.cmo \
unit_parsing_cpp.cmi
unit_parsing_cpp.cmx : parse_cpp.cmx ../../commons/oUnit.cmx \
flag_parsing_cpp.cmx ../../globals/config_pfff.cmx \
../../commons/common2.cmx ../../commons/common.cmx ast_cpp.cmx \
unit_parsing_cpp.cmi
unit_parsing_cpp.cmi : ../../commons/oUnit.cmi
visitor_cpp.cmo : ../../commons/ocaml.cmi ast_cpp.cmo visitor_cpp.cmi
visitor_cpp.cmx : ../../commons/ocaml.cmx ast_cpp.cmx visitor_cpp.cmi
visitor_cpp.cmi : ast_cpp.cmo

4
lang_cpp/parsing/META Normal file
View file

@ -0,0 +1,4 @@
description = "C/C++ parser"
requires = "unix num"
archive(byte) = "lib.cma"
archive(native) = "lib.cmxa"

101
lang_cpp/parsing/Makefile Normal file
View file

@ -0,0 +1,101 @@
TOP=../..
##############################################################################
# Variables
##############################################################################
TARGET=lib
-include $(TOP)/Makefile.config
SRC= flag_parsing_cpp.ml \
token_cpp.ml ast_cpp.ml \
type_cpp.ml \
meta_ast_cpp.ml \
visitor_cpp.ml lib_parsing_cpp.ml \
parser_cpp_mly_helper.ml parser_cpp.ml lexer_cpp.ml \
token_helpers_cpp.ml token_views_cpp.ml token_views_context.ml \
parsing_hacks_lib.ml pp_token.ml \
parsing_hacks_pp.ml parsing_hacks_cpp.ml parsing_hacks_typedef.ml \
parsing_hacks_define.ml \
parsing_hacks.ml \
parsing_recovery_cpp.ml \
parse_cpp.ml \
test_dump_nim.ml \
test_parsing_cpp.ml unit_parsing_cpp.ml
SYSLIBS= str.cma unix.cma
LIBS=$(TOP)/commons/lib.cma \
$(TOP)/h_program-lang/lib.cma
INCLUDEDIRS= \
$(TOP)/commons \
$(TOP)/commons_core \
$(TOP)/globals \
$(TOP)/h_program-lang
##############################################################################
# Generic variables
##############################################################################
-include $(TOP)/Makefile.common
##############################################################################
# Top rules
##############################################################################
all:: $(TARGET).cma
all.opt:: $(TARGET).cmxa
$(TARGET).cma: $(OBJS)
$(OCAMLC) -a -o $(TARGET).cma $(OBJS)
$(TARGET).cmxa: $(OPTOBJS) $(LIBS:.cma=.cmxa)
$(OCAMLOPT) -a -o $(TARGET).cmxa $(OPTOBJS)
$(TARGET).top: $(OBJS) $(LIBS)
$(OCAMLMKTOP) -o $(TARGET).top $(SYSLIBS) $(LIBS) $(OBJS)
clean::
rm -f $(TARGET).top
lexer_cpp.ml: lexer_cpp.mll
$(OCAMLLEX) $<
clean::
rm -f lexer_cpp.ml
beforedepend:: lexer_cpp.ml
parser_cpp.ml parser_cpp.mli: parser_cpp.mly
$(OCAMLYACC) $<
clean::
rm -f parser_cpp.ml parser_cpp.mli parser_cpp.output
beforedepend:: parser_cpp.ml parser_cpp.mli
visitor_cpp.cmo: visitor_cpp.ml
$(OCAMLC) -w y -c $<
parsing_hacks.cmo: parsing_hacks.ml
$(OCAMLC) -w -9 -c $<
parsing_hacks_cpp.cmo: parsing_hacks_cpp.ml
$(OCAMLC) -w -9 -c $<
parsing_hacks_pp.cmo: parsing_hacks_pp.ml
$(OCAMLC) -w -9 -c $<
parsing_hacks_typedef.cmo: parsing_hacks_typedef.ml
$(OCAMLC) -w -9 -c $<
token_views_context.cmo: token_views_context.ml
$(OCAMLC) -w -9 -c $<
##############################################################################
# install
##############################################################################
LIBNAME=pfff-lang_cpp
EXPORTSRC=meta_ast_cpp.mli \
parser_cpp.mli parse_cpp.mli \
lib_parsing_cpp.mli visitor_cpp.mli \
install-findlib:
ocamlfind install $(LIBNAME) META lib.cma lib.cmxa lib.a \
$(EXPORTSRC) $(EXPORTSRC:%.mli=%.cmi) \
ast_cpp.ml ast_cpp.cmi

831
lang_cpp/parsing/ast_cpp.ml Normal file
View file

@ -0,0 +1,831 @@
(* Yoann Padioleau
*
* Copyright (C) 2010-2014 Facebook
* Copyright (C) 2008-2009 University of Urbana Champaign
* Copyright (C) 2006-2007 Ecole des Mines de Nantes
* Copyright (C) 2002 Yoann Padioleau
*
* This program is free software; you can redistribute it and/or
* modify it under the terms of the GNU General Public License (GPL)
* version 2 as published by the Free Software Foundation.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* file license.txt for more details.
*)
(*****************************************************************************)
(* Prelude *)
(*****************************************************************************)
(*
* This is a big file ... C++ is a big and complicated language ...
* This file started with a simple AST for C. It was then extended
* to deal with cpp idioms (see 'cppext:' tag), gcc extensions (see gccext),
* and finally C++ constructs (see c++ext). A few kencc extensions
* were also recently added (see kenccext).
*
* gcc introduced StatementExpr which made expr and statement mutually
* recursive. It also added NestedFunc for even more mutual recursivity ...
* With C++ templates, because template arguments can be types or expressions
* and because templates are also qualifiers, almost all types
* are now mutually recursive ...
*
* Like most other ASTs in pfff, it's actually more a Concrete Syntax Tree.
* Some stuff are tagged 'semantic:' which means that they are computed
* after parsing.
*
* See also lang_c/parsing/ast_c.ml and lang_clang/parsing/ast_clang.ml
* (as well as mini/ast_minic.ml).
*
* todo:
* - migrate everything to wrap2, e.g. no more expressionbis, statementbis
* - support C++0x11, e.g. lambdas
*
* related work:
* - https://github.com/facebook/facebook-clang-plugins
* or https://github.com/Antique-team/clangml
* but by both using clang they work after preprocessing. This is
* fine for bug finding, but for codemap we need to parse as is,
* and we need to do it fast (calling clang is super expensive because
* calling cpp and parsing the end result is expensive)
* - EDG
* - see the CC'09 paper
*)
(*****************************************************************************)
(* The AST C++ related types *)
(*****************************************************************************)
(* ------------------------------------------------------------------------- *)
(* Token/info *)
(* ------------------------------------------------------------------------- *)
type tok = Parse_info.info
(* a shortcut to annotate some information with token/position information *)
and 'a wrap = 'a * tok list (* TODO: change to 'a * tok *)
and 'a wrap2 = 'a * tok
and 'a paren = tok * 'a * tok
and 'a brace = tok * 'a * tok
and 'a bracket = tok * 'a * tok
and 'a angle = tok * 'a * tok
and 'a comma_list = 'a wrap list
and 'a comma_list2 = ('a, tok (* the comma *)) Common.either list
(* with tarzan *)
(* ------------------------------------------------------------------------- *)
(* Ident, name, scope qualifier *)
(* ------------------------------------------------------------------------- *)
(* c++ext: in C 'name' and 'ident' are equivalent and are just strings.
* In C++ 'name' can have a complex form like 'A::B::list<int>::size'.
* I use Q for qualified. I also have a special type to make the difference
* between intermediate idents (the classname or template_id) and final idents.
* Note that sometimes final idents are also classnames and can have final
* template_id.
*
* Sometimes some elements are not allowed at certain places, for instance
* converters can not have an associated Qtop. But I prefered to simplify
* and have a unique type for all those different kinds of names.
*)
type name = tok (*::*) option * (qualifier * tok (*::*)) list * ident
and ident =
(* function name, macro name, variable, classname, enumname, namespace *)
| IdIdent of simple_ident
(* c++ext: *)
| IdTemplateId of simple_ident * template_arguments
| IdDestructor of tok(*~*) * simple_ident
| IdOperator of tok * (operator * tok list)
| IdConverter of tok * fullType
and simple_ident = string wrap2
and template_arguments = template_argument comma_list angle
and template_argument = (fullType, expression) Common.either
and qualifier =
| QClassname of simple_ident (* classname or namespacename *)
| QTemplateId of simple_ident * template_arguments
(* special cases *)
and class_name = name (* only IdIdent or IdTemplateId *)
and namespace_name = name (* only IdIdent *)
and typedef_name = name (* only IdIdent *)
and enum_name = name (* only IdIdent *)
and ident_name = name (* only IdIdent *)
(* TODO: do like in parsing_c/
* and ident_string =
* | RegularName of string wrap
*
* (* cppext: *)
* | CppConcatenatedName of (string wrap) wrap2 (* the ## separators *) list
* (* normally only used inside list of things, as in parameters or arguments
* * in which case, cf cpp-manual, it has a special meaning *)
* | CppVariadicName of string wrap (* ## s *)
* | CppIdentBuilder of string wrap (* s ( ) *) *
* ((string wrap) wrap2 list) (* arguments *)
*)
(* ------------------------------------------------------------------------- *)
(* Types *)
(* ------------------------------------------------------------------------- *)
(* We could have a more precise type in fullType, in expression, etc, but
* it would require too much things at parsing time such as checking whether
* there is no conflicts structname, computing value, etc. It's better to
* separate concerns, so I put '=>' to mean what we would really like. In fact
* what we really like is defining another fullType, expression, etc
* from scratch, because many stuff are just sugar.
*
* invariant: Array and FunctionType have also typeQualifier but they
* dont have sense. I put this to factorise some code. If you look in
* grammar, you see that we can never specify const for the array
* himself (but we can do it for pointer).
*)
and fullType = typeQualifier * typeC
and typeC = typeCbis wrap
(* less: rename to TBase, TPointer, etc *)
and typeCbis =
| BaseType of baseType
| Pointer of (* '*' *) fullType
(* c++ext: *)
| Reference of (* '&' *) fullType
| Array of constExpression option bracket * fullType
| FunctionType of functionType
| EnumName of tok (* 'enum' *) * simple_ident (*enum_name*)
| StructUnionName of structUnion wrap2 * simple_ident (*ident_name*)
(* c++ext: TypeName can now correspond also to a classname or enumname
* and is a name so can have some IdTemplateId in it.
*)
| TypeName of name(*typedef_name*)
(* only to disambiguate I think *)
| TypenameKwd of tok (* 'typename' *) * name(*typedef_name*)
(* gccext: TypeOfType may seems useless, why declare a __typeof__(int)
* x; ? But when used with macro, it allows to fix a problem of C which
* is that type declaration can be spread around the ident. Indeed it
* may be difficult to have a macro such as '#define macro(type,
* ident) type ident;' because when you want to do a macro(char[256],
* x), then it will generate invalid code, but with a '#define
* macro(type, ident) __typeof(type) ident;' it will work. *)
| TypeOf of tok * (fullType, expression) Common.either paren
(* should be really just at toplevel *)
| EnumDef of enum_definition (* => string * int list *)
(* c++ext: bigger type now *)
| StructDef of class_definition
(* forunparser: *)
| ParenType of fullType paren
and baseType =
| Void
| IntType of intType
| FloatType of floatType
(* stdC: type section. 'char' and 'signed char' are different *)
and intType =
| CChar (* obsolete? | CWchar *)
| Si of signed
(* c++ext: maybe could be put in baseType instead ? *)
| CBool | WChar_t
and signed = sign * base
and base =
| CChar2 | CShort | CInt | CLong
(* gccext: *)
| CLongLong
and sign = Signed | UnSigned
and floatType = CFloat | CDouble | CLongDouble
and typeQualifier =
{ const: tok option; volatile: tok option; }
(* TODO: like in parsing_c/
* (* gccext: cppext: *)
* and attribute = attributebis wrap
* and attributebis =
* | Attribute of string
*)
(* ------------------------------------------------------------------------- *)
(* Expressions *)
(* ------------------------------------------------------------------------- *)
(* Because of StatementExpr, we can have more 'new scope', but it's
* rare I think. For instance with 'array of constExpression' we could
* have an StatementExpr and a new (local) struct defined. Same for
* Constructor.
*)
and expression = expressionbis wrap
and expressionbis =
(* Id can be an enumeration constant, variable, function name.
* cppext: Id can also be the name of a macro. sparse says
* "an identifier with a meaning is a symbol".
* c++ext: Id is now a 'name' instead of a 'string' and can be
* also an operator name.
*)
| Id of name * ident_info (* semantic: see check_variables_cpp.ml *)
| C of constant
(* I used to have FunCallSimple but not that useful, and we want scope info
* for FunCallSimple too because can have fn(...) where fn is actually
* a local *)
| Call of expression * argument comma_list paren
(* gccext: x ? /* empty */ : y <=> x ? x : y; *)
| CondExpr of expression * expression option * expression
(* should be considered as statements, bad C langage *)
| Sequence of expression * expression
| Assignment of expression * assignOp * expression
| Postfix of expression * fixOp
| Infix of expression * fixOp
(* contains GetRef and Deref!! todo: lift up? *)
| Unary of expression * unaryOp
| Binary of expression * binaryOp * expression
| ArrayAccess of expression * expression bracket
(* The Pt is redundant normally, could be replace by DeRef RecordAccess *)
| RecordAccess of expression * name
| RecordPtAccess of expression * name
(* c++ext: note that second paramater is an expression, not a name *)
| RecordStarAccess of expression * expression
| RecordPtStarAccess of expression * expression
| SizeOfExpr of tok * expression
| SizeOfType of tok * fullType paren
| Cast of fullType paren * expression
(* gccext: *)
| StatementExpr of compound paren (* ( { } ) new scope*)
(* gccext: kenccext: *)
| GccConstructor of fullType paren * initialiser comma_list brace
(* c++ext: *)
| This of tok
| ConstructedObject of fullType * argument comma_list paren
| TypeId of tok * (fullType, expression) Common.either paren
| CplusplusCast of cast_operator wrap2 * fullType angle * expression paren
| New of tok (*::*) option * tok *
argument comma_list paren option (* placement *) *
fullType *
argument comma_list paren option (* initializer *)
| Delete of tok (*::*) option * expression
| DeleteArray of tok (*::*) option * expression
| Throw of expression option
(* forunparser: *)
| ParenExpr of expression paren
| ExprTodo
(* see check_variables_cpp.ml *)
and ident_info = {
mutable i_scope: Scope_code.scope;
}
(* cppext: normmally just expression *)
and argument = (expression, weird_argument) Common.either
and weird_argument =
| ArgType of fullType
(* for really unparsable stuff ... we just bailout *)
| ArgAction of action_macro
and action_macro =
| ActMisc of tok list
(* I put 'string' for Int and Float because 'int' would not be enough.
* Indeed OCaml int are 31 bits. So it's simpler to use 'string'.
* Same reason to have 'string' instead of 'int list' for the String case.
*
* note: '-2' is not a constant; it is the unary operator '-'
* applied to the constant '2'. So the string must represent a positive
* integer only.
*)
and constant =
| Int of (string (* * intType*))
| Float of (string * floatType)
| Char of (string * isWchar) (* normally it is equivalent to Int *)
| String of (string * isWchar)
| MultiString (* can contain MacroString *)
(* c++ext: *)
| Bool of bool
and isWchar = IsWchar | IsChar
and unaryOp =
(* less: could be lift up, those are really important operators *)
| GetRef | DeRef
(* gccext: via &&label notation *)
| GetRefLabel
| UnPlus | UnMinus | Tilde | Not
and assignOp = SimpleAssign | OpAssign of arithOp
and fixOp = Dec | Inc
and binaryOp = Arith of arithOp | Logical of logicalOp
and arithOp =
| Plus | Minus | Mul | Div | Mod
| DecLeft | DecRight
| And | Or | Xor
and logicalOp =
| Inf | Sup | InfEq | SupEq
| Eq | NotEq
| AndLog | OrLog
(* c++ext: used elsewhere but prefer to define it close to other operators *)
and ptrOp = PtrStarOp | PtrOp
and allocOp = NewOp | DeleteOp | NewArrayOp | DeleteArrayOp
and accessop = ParenOp | ArrayOp
and operator =
| BinaryOp of binaryOp
| AssignOp of assignOp
| FixOp of fixOp
| PtrOpOp of ptrOp
| AccessOp of accessop
| AllocOp of allocOp
| UnaryTildeOp | UnaryNotOp | CommaOp
(* c++ext: *)
and cast_operator =
| Static_cast | Dynamic_cast | Const_cast | Reinterpret_cast
and constExpression = expression (* => int *)
(* ------------------------------------------------------------------------- *)
(* Statements *)
(* ------------------------------------------------------------------------- *)
(* note: assignement is not a statement, it's an expression :(
* (wonderful C language).
* note: I use 'and' for type definition because gccext allows statements as
* expressions, so we need mutual recursive type definition now.
*)
and statement = statementbis wrap
and statementbis =
| Compound of compound (* new scope *)
| ExprStatement of exprStatement
| Labeled of labeled
| Selection of selection
| Iteration of iteration
| Jump of jump
(* c++ext: in C this constructor could be outside the statement type, in a
* decl type, because declarations are only at the beginning of a compound
* normally. But in C++ we can freely mix declarations and statements.
*)
| DeclStmt of block_declaration
(* c++ext: *)
| Try of tok * compound * handler list
(* gccext: *)
| NestedFunc of func_definition
(* cppext: *)
| MacroStmt
| StmtTodo
(* cppext: c++ext:
* old: compound = (declaration list * statement list)
* old: (declaration, statement) either list
*)
and compound = statement_sequencable list brace
and exprStatement = expression option
and labeled =
| Label of string * statement
| Case of expression * statement
| CaseRange of expression * expression * statement (* gccext: *)
| Default of statement
and selection =
| If of tok * expression paren * statement * tok option * statement
(* need to check that all elements in the compound start
* with a case:, otherwise it's unreachable code.
*)
| Switch of tok * expression paren * statement
and iteration =
| While of tok * expression paren * statement
| DoWhile of tok * statement * tok * expression paren * tok (*;*)
| For of
tok *
(exprStatement wrap * exprStatement wrap * exprStatement wrap) paren *
statement
(* cppext: *)
| MacroIteration of simple_ident * argument comma_list paren * statement
and jump =
| Goto of string
| Continue | Break
| Return | ReturnExpr of expression
(* gccext: goto *exp ';' *)
| GotoComputed of expression
(* c++ext: *)
and handler = tok * exception_declaration paren * compound
and exception_declaration =
| ExnDeclEllipsis of tok
| ExnDecl of parameter
(* easier to put at statement_list level than statement level *)
and statement_sequencable =
| StmtElem of statement
(* cppext: *)
| CppDirectiveStmt of cpp_directive
| IfdefStmt of ifdef_directive (* * statement list *)
(* ------------------------------------------------------------------------- *)
(* Block Declaration *)
(* ------------------------------------------------------------------------- *)
(* a.k.a declaration_statement *)
and block_declaration =
(* Before I had a Typedef constructor, but why make this special case and not
* have also StructDef, EnumDef, so that 'struct t {...} v' which would
* then generate two declarations. If you want a cleaner C AST use
* ast_c.ml.
* note: before the need for unparser, I didn't have a DeclList but just
* a Decl.
*)
| DeclList of onedecl comma_list * tok (*;*)
(* cppext: todo? now factorize with MacroTop ? *)
| MacroDecl of tok list * simple_ident * argument comma_list paren * tok
(* c++ext: using namespace *)
| UsingDecl of (tok * name * tok (*;*))
| UsingDirective of tok * tok (*'namespace'*) * namespace_name * tok(*;*)
| NameSpaceAlias of tok * simple_ident * tok (*=*) * namespace_name * tok(*;*)
(* gccext: *)
| Asm of tok * tok option (*volatile*) * asmbody paren * tok(*;*)
(* gccext: *)
and asmbody = tok list (* string list *) * colon wrap (* : *) list
and colon = Colon of colon_option comma_list
and colon_option = colon_optionbis wrap
and colon_optionbis = ColonMisc | ColonExpr of expression paren
(* ------------------------------------------------------------------------- *)
(* Variable definition (and also field definition) *)
(* ------------------------------------------------------------------------- *)
(* note: onedecl includes prototype declarations and class_declarations!
* c++ext: onedecl now covers also field definitions as fields can have
* storage in C++.
*)
and onedecl = {
(* option cos can have empty declaration or struct tag declaration.
* kenccext: name can also be empty because of anonymous fields.
*)
v_namei: (name * init option) option;
v_type: fullType;
v_storage: storage;
(* v_attr: attribute list; *) (* gccext: *)
}
and storage = NoSto | StoTypedef of tok | Sto of storageClass wrap2
and storageClass = Auto | Static | Register | Extern
(* Friend ???? Mutable? *)
(*c++ext: TODO *)
(* I am not sure what it means to declare a prototype inline, but gcc
* accepts it. *)
and _func_specifier = Inline | Virtual
and init =
| EqInit of tok (*=*) * initialiser
(* c++ext: constructed object *)
| ObjInit of argument comma_list paren
and initialiser =
| InitExpr of expression
| InitList of initialiser comma_list brace
(* gccext: *)
| InitDesignators of designator list * tok (*=*) * initialiser
| InitFieldOld of simple_ident * tok (*:*) * initialiser
| InitIndexOld of expression bracket * initialiser
(* ex: [2].y = x, or .y[2] or .y.x. They can be nested *)
and designator =
| DesignatorField of tok(*:*) * simple_ident
| DesignatorIndex of expression bracket
| DesignatorRange of (expression * tok (*...*) * expression) bracket
(* ------------------------------------------------------------------------- *)
(* Function definition *)
(* ------------------------------------------------------------------------- *)
(* Normally we should define another type functionType2 because there
* are more restrictions on what can define a function than a pointer
* function. For instance a function declaration can omit the name of the
* parameter wheras a function definition can not. But, in some cases such
* as 'f(void) {', there is no name too, so I simplified and reused the
* same functionType type for both declarations and function definitions.
*)
and func_definition = {
f_name: name;
f_type: functionType;
f_storage: storage;
(* todo: gccext: inline or not:, f_inline: tok option *)
f_body: compound;
(*f_attr: attribute list;*) (* gccext: *)
}
and functionType = {
ft_ret: fullType; (* fake return type for ctor/dtor *)
ft_params: parameter comma_list paren;
ft_dots: (tok(*,*) * tok(*...*)) option;
(* c++ext: *)
ft_const: tok option; (* only for methods *)
ft_throw: exn_spec option;
}
and parameter = {
p_name: simple_ident option;
p_type: fullType;
p_register: tok option;
(* c++ext: *)
p_val: (tok (*=*) * expression) option;
}
and exn_spec = (tok * name comma_list2 paren)
(* less: simplify? need differentiate at this level? could have
* is_ctor, is_dtor helper instead.
*)
and func_or_else =
| FunctionOrMethod of func_definition
(* c++ext: special member function *)
| Constructor of func_definition (* TODO explicit/inline, chain_call *)
| Destructor of func_definition
and method_decl =
| MethodDecl of onedecl * (tok * tok) option (* '=' '0' *) * tok(*;*)
| ConstructorDecl of
simple_ident * parameter comma_list paren * tok(*;*)
| DestructorDecl of
tok(*~*) * simple_ident * tok option paren * exn_spec option * tok(*;*)
(* ------------------------------------------------------------------------- *)
(* enum definition *)
(* ------------------------------------------------------------------------- *)
(* less: use a record *)
and enum_definition =
tok (*enum*) * simple_ident option * enum_elem comma_list brace
and enum_elem = {
e_name: simple_ident;
e_val: (tok (*=*) * constExpression) option;
}
(* ------------------------------------------------------------------------- *)
(* Class definition *)
(* ------------------------------------------------------------------------- *)
and class_definition = {
c_kind: structUnion wrap2;
(* the ident can be a template_id when do template specialization. *)
c_name: ident_name(*class_name??*) option;
(* c++ext: *)
c_inherit: (tok (* ':' *) * base_clause comma_list) option;
c_members: class_member_sequencable list brace (* new scope *);
}
and structUnion =
| Struct | Union
(* c++ext: *)
| Class
and base_clause = {
i_name: class_name;
i_virtual: tok option;
i_access: access_spec wrap2 option;
}
(* used in inheritance spec (base_clause) and class_member *)
and access_spec = Public | Private | Protected
(* was called field wrap before *)
and class_member =
(* could put outside and take class_member list *)
| Access of access_spec wrap2 * tok (*:*)
(* before unparser, I didn't have a FieldDeclList but just a Field. *)
| MemberField of fieldkind comma_list * tok (*';'*)
| MemberFunc of func_or_else
| MemberDecl of method_decl
| QualifiedIdInClass of name (* ?? *) * tok(*;*)
| TemplateDeclInClass of (tok * template_parameters * declaration)
| UsingDeclInClass of (tok (*using*) * name * tok (*;*))
(* gccext: and maybe c++ext: *)
| EmptyField of tok (*;*)
(* At first I thought that a bitfield could be only Signed/Unsigned.
* But it seems that gcc allow char i:4. C rule must say that you
* can cast into int so enum too, ...
* c++ext: FieldDecl was before Simple of string option * fullType
* but in c++ fields can also have storage (e.g. static) so now reuse
* ondecl.
*)
and fieldkind =
| FieldDecl of onedecl
| BitField of simple_ident option * tok(*:*) *
fullType * constExpression
(* fullType => BitFieldInt | BitFieldUnsigned *)
and class_member_sequencable =
| ClassElem of class_member
(* cppext: *)
| CppDirectiveStruct of cpp_directive
| IfdefStruct of ifdef_directive (* * field list *)
(* ------------------------------------------------------------------------- *)
(* cppext: cpp directives, #ifdef, #define and #include body *)
(* ------------------------------------------------------------------------- *)
and cpp_directive =
| Define of tok (* #define*) * simple_ident * define_kind * define_val
| Include of tok (* #include s *) * inc_kind * string (* path *)
| Undef of simple_ident (* #undef xxx *)
| PragmaAndCo of tok
and define_kind =
| DefineVar
| DefineFunc of string wrap comma_list paren
and define_val =
| DefineExpr of expression
| DefineStmt of statement
| DefineType of fullType
| DefineFunction of func_definition
| DefineInit of initialiser (* in practice only { } with possible ',' *)
(* ?? *)
| DefineText of string wrap
| DefineEmpty
| DefineDoWhileZero of statement wrap (* do { } while(0) *)
| DefinePrintWrapper of tok (* if *) * expression paren * name
| DefineTodo
and inc_kind =
| Local (* "" *)
| Standard (* <> *)
| Weird (* ex: #include SYSTEM_H *)
(* less: 'a ifdefed = 'a list wrap (* ifdef elsif else endif *) *)
and ifdef_directive = ifdefkind wrap2
and ifdefkind =
| Ifdef (* todo? of string? *)
(* less: IfIf of formula_cpp ? *)
| IfdefElse
| IfdefElseif
| IfdefEndif
(* less:
* set in Parsing_hacks.set_ifdef_parenthize_info. It internally use
* a global so it means if you parse the same file twice you may get
* different id. I try now to avoid this pb by resetting it each
* time I parse a file.
*
* and matching_tag =
* IfdefTag of (int (* tag *) * int (* total with this tag *))
*)
(* ------------------------------------------------------------------------- *)
(* The toplevel elements *)
(* ------------------------------------------------------------------------- *)
(* it's not really 'toplevel' because the elements below can be nested
* inside namespaces or some extern. It's not really 'declaration'
* either because it can defines stuff. But I keep the C++ standard
* terminology.
*
* note that we use 'block_declaration' below, not 'statement'.
*)
and declaration =
| BlockDecl of block_declaration (* include struct/globals/... definitions *)
| Func of func_or_else
(* c++ext: *)
| TemplateDecl of tok * template_parameters * declaration
| TemplateSpecialization of tok * unit angle * declaration
(* the list can be empty *)
| ExternC of tok * tok * declaration
| ExternCList of tok * tok * declaration_sequencable list brace
(* the list can be empty *)
| NameSpace of tok * simple_ident * declaration_sequencable list brace
(* after have some semantic info *)
| NameSpaceExtend of string * declaration_sequencable list
| NameSpaceAnon of tok * declaration_sequencable list brace
(* gccext: allow redundant ';' *)
| EmptyDef of tok
| DeclTodo
(* c++ext: *)
and template_parameter = parameter (* todo? more? *)
and template_parameters = template_parameter comma_list angle
(* easier to put at statement_list level than statement level *)
and declaration_sequencable =
| DeclElem of declaration
(* cppext: *)
| CppDirectiveDecl of cpp_directive
| IfdefDecl of ifdef_directive (* * toplevel list *)
(* cppext: *)
| MacroTop of simple_ident * argument comma_list paren * tok option
| MacroVarTop of simple_ident * tok (* ; *)
(* could also be in decl *)
| NotParsedCorrectly of tok list
and toplevel = declaration_sequencable
and program = toplevel list
(* ------------------------------------------------------------------------- *)
(* Any *)
(* ------------------------------------------------------------------------- *)
and any =
| Program of program
| Toplevel of toplevel
| Cpp of cpp_directive
| Stmt of statement
| Expr of expression
| Type of fullType
| Name of name
| BlockDecl2 of block_declaration
| ClassDef of class_definition
| FuncDef of func_definition
| FuncOrElse of func_or_else
| ClassMember of class_member
| OneDecl of onedecl
| Init of initialiser
| Constant of constant
| Argument of argument
| Parameter of parameter
| Body of compound
| Info of tok
| InfoList of tok list
(* with tarzan *)
(*****************************************************************************)
(* Some constructors *)
(*****************************************************************************)
let nQ = {const=None; volatile= None}
let noIdInfo () = { i_scope = Scope_code.NoScope; }
let noii = []
let noQscope = []
(*****************************************************************************)
(* Wrappers *)
(*****************************************************************************)
let unwrap x = fst x
let uncomma xs = List.map fst xs
let unparen (_, x, _) = x
let unbrace (_, x, _) = x
let unwrap_typeC (_qu, (typeC, _ii)) = typeC
(* When want add some info in AST that does not correspond to
* an existing C element.
* old: when don't want 'synchronize' on it in unparse_c.ml
* (now have other mark for tha matter).
* used by parsing hacks
*)
let make_expanded ii =
let noVirtPos = ({Parse_info.str="";charpos=0;line=0;column=0;file=""},-1) in
let (a, b) = noVirtPos in
{ ii with Parse_info.token = Parse_info.ExpandedTok
(Parse_info.get_original_token_location ii.Parse_info.token, a, b) }
(* used by parsing hacks *)
let rewrap_pinfo pi ii =
{ii with Parse_info.token = pi}
(* used while migrating the use of 'string' to 'name' in check_variables *)
let (string_of_name_tmp: name -> string) = fun name ->
let (_opt, _qu, id) = name in
match id with
| IdIdent (s,_) -> s
| _ -> failwith "TODO:string_of_name_tmp"
let (ii_of_id_name: name -> tok list) = fun name ->
let (_opt, _qu, id) = name in
match id with
| IdIdent (_s,ii) -> [ii]
| IdOperator (_, (_op, ii)) -> ii
| IdConverter (_tok, _ft) -> failwith "ii_of_id_name: IdConverter"
| IdDestructor (tok, (_s, ii)) -> [tok;ii]
| IdTemplateId ((_s, ii), _args) -> [ii]

View file

@ -0,0 +1,2 @@
Yoann Padioleau <pad@fb.com>

View file

@ -0,0 +1,93 @@
-*- org -*-
TODO:
http://blog.robertelder.org/jim-roskind-grammar/
http://eli.thegreenplace.net/2007/11/24/the-context-sensitivity-of-cs-grammar/
* Typedefs
simple_type_specifier:
...
| type_cplusplus_id { Right3 (TypeName $1), noii }
(* history: cant put TIdent {} cos it makes the grammar ambiguous and
* generates lots of conflicts => we must use some tricks.
* We used make the lexer and parser cooperate (in a lexerParser.ml file).
* But this was not enough because of declarations such as 'acpi acpi;'
* and so we had to enable/disable the ident->typedef mechanism
* (which requires even more lexer/parser cooperation). But
* this was ugly too so now we use a typedef "inference" mechanism.
We do many things to handle typedefs ambiguities:
- parsing_hack_typedef heuristics
- token_view_context in Parameter heuristics
- dealing with template before the actual typedef
- a few rules added for parameter and arguments to allow
both TIdent and TIdent_typedef in both contexts
- ...
** pointer decl, multiplication and ambiguity
from "Yacc Is Dead" at http://lambda-the-ultimate.org/node/4148#comment
" 'x*y;' in C++ this could be a multiplication, pointer declaration, or
arbitrary overloaded meaning of "*". You have to hit name and type
resolution before you can distinguish them."
** cast and ambiguity
can be cast or binary minus
(u32int)-pa
can be cast or funcall
(u32int)(-pa));
same with (uintptr)&x.
* If-then-else
see dangling-else section in lang_php/parsing/conflicts.txt
* Template < >
We do many things ...
* C++
* TODO ':'
When have 'class X :' we don't know if it's the start of possibly a
class with inheritance spec, or a bitfield as 'class X :3'.
TODO why have conflict on TCol ???
* Old notes
(* Cocci: Each token will be decorated in the future by the mcodekind
* of cocci. It is the job of the pretty printer to look at this
* information and decide to print or not the token (and also the
* pending '+' associated sometimes with the token).
*
* The first time that we parse the original C file, the mcodekind is
* empty, or more precisely all is tagged as a CONTEXT with NOTHING
* associated. This is what I call a "clean" expr/statement/....
*
* Each token will also be decorated in the future with an environment,
* because the pending '+' may contain metavariables that refer to some
* C code.
*
* Update: Now I use a ref! so take care.
*
* Sometimes we want to add someting at the beginning or at the end
* of a construct. For 'function' and 'decl' we want add something
* to their left and for 'if' 'while' et 'for' and so on at their right.
* We want some kinds of "virtual placeholders" that represent the start or
* end of a construct. We use fakeInfo for that purpose.
* To identify those cases I have added a fakestart/fakeend comment.
*
* convention: I often use 'ii' for the name of a list of info.
*
*)

View file

@ -0,0 +1,14 @@
parsing_c++ library - Yoann Padioleau
Copyright (C) 2002-2008 Yoann Padioleau, University of Urbana Champaign,
Ecole des Mines de Nantes, Universite de Rennes 1.
This program is free software; you can redistribute it and/or
modify it under the terms of the GNU General Public License (GPL)
version 2 as published by the Free Software Foundation.
This program is distributed in the hope that it will be useful,
but WITHOUT ANY WARRANTY; without even the implied warranty of
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
file license.txt for more details.

View file

@ -0,0 +1,9 @@
Thanks to Julia Lawall for the idea to better parse C+CPP by using
indentation information for heuristic-based parsing. Thanks to Julia
again for many other things too long to enumerate.
Inspiration:
- C yacc grammar published in 1985 by Jeff Lee:
lex: http://www.lysator.liu.se/c/ANSI-C-grammar-l.html
yacc: http://www.lysator.liu.se/c/ANSI-C-grammar-y.html

View file

@ -0,0 +1,77 @@
(*****************************************************************************)
(* types *)
(*****************************************************************************)
type language =
| C
| Cplusplus
(*****************************************************************************)
(* macros *)
(*****************************************************************************)
let macros_h =
ref (Filename.concat Config_pfff.path "/data/cpp_stdlib/macros.h")
let cmdline_flags_macrofile () = [
"-macros", Arg.Set_string macros_h,
" <file>";
]
(*****************************************************************************)
(* verbose *)
(*****************************************************************************)
let verbose_lexing = ref true
let verbose_parsing = ref true
(* do not raise Parse_error in parse_cpp.ml, try to recover! *)
let error_recovery = ref true
let show_parsing_error = ref true
let verbose_pp_ast = ref false
let filter_msg = ref false
let filter_classic_passed = ref false
let filter_define_error = ref true
let cmdline_flags_verbose () = [
"-verbose_parsing_cpp", Arg.Set verbose_parsing, " ";
]
(*****************************************************************************)
(* debugging *)
(*****************************************************************************)
let debug_lexer = ref false
let debug_typedef = ref false
let debug_pp = ref false
let debug_pp_ast = ref false
let debug_cplusplus = ref false
let cmdline_flags_debugging () = [
"-debug_lexer_cpp", Arg.Set debug_lexer , " ";
"-debug_pp", Arg.Set debug_pp, " ";
"-debug_typedef", Arg.Set debug_typedef, " ";
"-debug_cplusplus", Arg.Set debug_cplusplus, " ";
"-debug_cpp", Arg.Unit (fun () ->
debug_pp := true;
debug_typedef := true;
debug_cplusplus := true;
), " ";
]
(*****************************************************************************)
(* Disable parsing features *)
(*****************************************************************************)
let strict_lexer = ref false
let if0_passing = ref true
let ifdef_to_if = ref false
let sgrep_mode = ref false

View file

@ -0,0 +1,710 @@
{
(* Yoann Padioleau
*
* Copyright (C) 2002 Yoann Padioleau
* Copyright (C) 2006-2007 Ecole des Mines de Nantes
* Copyright (C) 2008-2009 University of Urbana Champaign
* Copyright (C) 2010-2013 Facebook
*
* This program is free software; you can redistribute it and/or
* modify it under the terms of the GNU General Public License (GPL)
* version 2 as published by the Free Software Foundation.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* file license.txt for more details.
*)
open Common
open Parser_cpp
open Ast_cpp (* to factorise tokens with OpAssign, ... *)
module Flag = Flag_parsing_cpp
module Ast = Ast_cpp
module PI = Parse_info
(*****************************************************************************)
(* Prelude *)
(*****************************************************************************)
(* The C/cpp/C++ lexer.
*
* This lexer generates tokens for C (int, while, ...), C++ (new, delete, ...),
* CPP (#define, #ifdef, ...).
* It also generate tokens for comments and spaces. This means that
* it can not be used as-is. Some post-filtering
* has to be done to feed it to a parser. Note that C and C++ are not
* context free languages and so some idents must be disambiguated
* in some ways. TIdent below must thus be post-processed too (as well
* as other tokens like '<' for C++). See parsing_hack.ml for examples.
*
* note: We can't use Lexer_parser._lexer_hint here to do different
* things because we now call the lexer to get all the tokens
* and then only we parse. So we can use the hint only
* in parse_cpp.ml. For the same reason, we don't handle typedefs
* here anymore. We really just tokenize ...
*)
(*****************************************************************************)
(* Helpers *)
(*****************************************************************************)
exception Lexical of string
let error s =
if !Flag.strict_lexer
then raise (Lexical s)
else
if !Flag.verbose_lexing
then pr2 ("LEXER: " ^ s)
else ()
let tok lexbuf =
Lexing.lexeme lexbuf
let tokinfo lexbuf =
Parse_info.tokinfo_str_pos (tok lexbuf) (Lexing.lexeme_start lexbuf)
let tok_add_s = Parse_info.tok_add_s
(* ---------------------------------------------------------------------- *)
(* Keywords *)
(* ---------------------------------------------------------------------- *)
(* opti: less convenient, but using a hash is faster than using a match *)
let keyword_table = Common.hash_of_list [
(* c: *)
"void", (fun ii -> Tvoid ii);
"char", (fun ii -> Tchar ii);
"short", (fun ii -> Tshort ii); "int", (fun ii -> Tint ii);
"long", (fun ii -> Tlong ii);
"float", (fun ii -> Tfloat ii); "double", (fun ii -> Tdouble ii);
"unsigned", (fun ii -> Tunsigned ii); "signed", (fun ii -> Tsigned ii);
"auto", (fun ii -> Tauto ii);
"register", (fun ii -> Tregister ii);
"extern", (fun ii -> Textern ii);
"static", (fun ii -> Tstatic ii);
"const", (fun ii -> Tconst ii); "volatile", (fun ii -> Tvolatile ii);
"struct", (fun ii -> Tstruct ii);
"union", (fun ii -> Tunion ii);
"enum", (fun ii -> Tenum ii);
"typedef", (fun ii -> Ttypedef ii);
"if", (fun ii -> Tif ii); "else", (fun ii -> Telse ii);
"break", (fun ii -> Tbreak ii); "continue", (fun ii -> Tcontinue ii);
"switch", (fun ii -> Tswitch ii);
"case", (fun ii -> Tcase ii); "default", (fun ii -> Tdefault ii);
"for", (fun ii -> Tfor ii);
"do", (fun ii -> Tdo ii);
"while", (fun ii -> Twhile ii);
"return", (fun ii -> Treturn ii);
"goto", (fun ii -> Tgoto ii);
"sizeof", (fun ii -> Tsizeof ii);
(* gccext: more (cpp) aliases are in macros.h *)
"asm", (fun ii -> Tasm ii);
"__attribute__", (fun ii -> Tattribute ii);
"typeof", (fun ii -> Ttypeof ii);
(* also a c++ext: *)
"inline", (fun ii -> Tinline ii);
(* c99: *)
"__restrict__", (fun ii -> Trestrict ii);
(* c++ext: see also TH.is_cpp_keyword *)
"class", (fun ii -> Tclass ii);
"this", (fun ii -> Tthis ii);
"new" , (fun ii -> Tnew ii);
"delete" , (fun ii -> Tdelete ii);
"template" , (fun ii -> Ttemplate ii);
"typeid" , (fun ii -> Ttypeid ii);
"typename" , (fun ii -> Ttypename ii);
"catch" , (fun ii -> Tcatch ii);
"try" , (fun ii -> Ttry ii);
"throw" , (fun ii -> Tthrow ii);
"operator", (fun ii -> Toperator ii);
"public" , (fun ii -> Tpublic ii);
"private" , (fun ii -> Tprivate ii);
"protected" , (fun ii -> Tprotected ii);
"friend" , (fun ii -> Tfriend ii);
"virtual", (fun ii -> Tvirtual ii);
"namespace", (fun ii -> Tnamespace ii);
"using", (fun ii -> Tusing ii);
"bool", (fun ii -> Tbool ii);
"true", (fun ii -> Ttrue ii); "false", (fun ii -> Tfalse ii);
"wchar_t", (fun ii -> Twchar_t ii);
"const_cast" , (fun ii -> Tconst_cast ii);
"dynamic_cast" , (fun ii -> Tdynamic_cast ii);
"static_cast" , (fun ii -> Tstatic_cast ii);
"reinterpret_cast" , (fun ii -> Treinterpret_cast ii);
"explicit", (fun ii -> Texplicit ii);
"mutable", (fun ii -> Tmutable ii);
"export", (fun ii -> Texport ii);
]
let error_radix s =
("numeric " ^ s ^ " constant contains digits beyond the radix:")
}
(*****************************************************************************)
(* Regexps aliases *)
(*****************************************************************************)
let letter = ['A'-'Z' 'a'-'z' '_']
let digit = ['0'-'9']
(* not used for the moment *)
let punctuation = ['!' '"' '#' '%' '&' '\'' '(' ')' '*' '+' ',' '-' '.' '/' ':'
';' '<' '=' '>' '?' '[' '\\' ']' '^' '{' '|' '}' '~']
let space = [' ' '\t' '\n' '\r' '\011' '\012' ]
let additionnal = [ ' ' '\b' '\t' '\011' '\n' '\r' '\007' ]
(* 7 = \a = bell in C. this is not the only char allowed !!
* ex @ and $ ` are valid too
*)
let cchar = (letter | digit | punctuation | additionnal)
let sp = [' ' '\t']+
let spopt = [' ' '\t']*
let dec = ['0'-'9']
let oct = ['0'-'7']
let hex = ['0'-'9' 'a'-'f' 'A'-'F']
let decimal = ('0' | (['1'-'9'] dec*))
let octal = ['0'] oct+
let hexa = ("0x" |"0X") hex+
let pent = dec+
let pfract = dec+
let sign = ['-' '+']
let exp = ['e''E'] sign? dec+
let real = pent exp | ((pent? '.' pfract | pent '.' pfract? ) exp?)
let id = letter (letter | digit) *
(*****************************************************************************)
(* Rule token *)
(*****************************************************************************)
rule token = parse
(* ----------------------------------------------------------------------- *)
(* Spaces, comments *)
(* ----------------------------------------------------------------------- *)
(* note: this lexer generate tokens for comments! So we can not give
* this lexer as-is to the parsing function. We must postprocess it, and
* use techniques like cur_tok ref in parse_cpp.ml
*)
| [' ' '\t' ]+
{ TCommentSpace (tokinfo lexbuf) }
(* see also TCppEscapedNewline below *)
| [ '\n' '\r' '\011' '\012']
{ TCommentNewline (tokinfo lexbuf) }
| "/*"
{ let info = tokinfo lexbuf in
let com = comment lexbuf in
TComment(info +> tok_add_s com)
}
(* C++ comments are allowed via gccext, but normally they are deleted by cpp.
* So we need this here only because we dont call cpp before.
* Note that we don't keep the trailing \n; it will be in another token.
*)
| "//" [^'\r' '\n' '\011']* { TComment (tokinfo lexbuf) }
(* ---------------------- *)
(* #include *)
(* ---------------------- *)
(* The difference between a local "" and standard <> include is computed
* later in parser_cpp.mly. So we redo a little bit of lexing there. It's
* ugly but simpler to generate a single token here. *)
| (("#" [' ''\t']* ("include" | "include_next" | "import")
[' ' '\t']*) as includes)
(('"' ([^ '"']+) '"' |
'<' [^ '>']+ '>' |
['A'-'Z''_']+
) as filename)
{ (* less: generate 2 info so highlight_cpp.ml can colorize the
* directive and the filename differently
*)
TInclude (includes, filename, tokinfo lexbuf)
}
(* ---------------------- *)
(* #ifdef *)
(* ---------------------- *)
| "#" [' ' '\t']* "if" [' ' '\t']* '0' (* [^'\n']* '\n' *)
{ let info = tokinfo lexbuf in
TIfdefBool (false, info(* +> tok_add_s (cpp_eat_until_nl lexbuf)*))
}
| "#" [' ' '\t']* "if" [' ' '\t']* '1' (* [^'\n']* '\n' *)
{ let info = tokinfo lexbuf in
TIfdefBool (true, info)
}
| "#" [' ' '\t']* "ifdef" [' ' '\t']* "__cplusplus" [^'\n']* '\n'
{ let info = tokinfo lexbuf in
TIfdefMisc (false, info)
}
(* can have some ifdef 0 hence the letter|digit even at beginning of word *)
| "#" [' ''\t']* "ifdef" [' ''\t']+ (letter|digit)((letter|digit)*) [' ''\t']*
{ TIfdef (tokinfo lexbuf) }
| "#" [' ''\t']* "ifndef" [' ''\t']+ (letter|digit)((letter|digit)*)[' ''\t']*
{ TIfdef (tokinfo lexbuf) }
| "#" [' ''\t']* "if" [' ' '\t']+
{ let info = tokinfo lexbuf in
TIfdef (info +> tok_add_s (cpp_eat_until_nl lexbuf))
}
| "#" [' ' '\t']* "if" '('
{ let info = tokinfo lexbuf in
TIfdef (info +> tok_add_s (cpp_eat_until_nl lexbuf))
}
| "#" [' ' '\t']* "elif" [' ' '\t']+
{ let info = tokinfo lexbuf in
TIfdefelif (info +> tok_add_s (cpp_eat_until_nl lexbuf))
}
(* bugfix: can have #endif LINUX but at the same time if I eat everything
* until next line, I may miss some TComment which for some tools
* are important such as aComment
*)
| "#" [' ' '\t']* "endif" (*[^'\n']* '\n'*)
{ TEndif (tokinfo lexbuf) }
| "#" [' ' '\t']* "else" [' ' '\t' '\n']
{ TIfdefelse (tokinfo lexbuf) }
(* ---------------------- *)
(* #define, #undef *)
(* ---------------------- *)
(* The rest of the lexing/parsing of #define is done in fix_tokens_define
* where we parse all TCppEscapedNewline and finally generate a TDefEol
*)
| "#" [' ' '\t']* "define" { TDefine (tokinfo lexbuf) }
(* note: in some cases we can have stuff after the ident as in #undef XXX 50,
* but I currently don't handle it cos I think it's bad code.
*)
| (("#" [' ' '\t']* "undef" [' ' '\t']+) as _undef) (id as id)
(* alt: +> tok_add_s (cpp_eat_until_nl lexbuf)) *)
{ TUndef (id, tokinfo lexbuf) }
(* ---------------------- *)
(* #define body *)
(* ---------------------- *)
(* We could generate separate tokens for #, ## and then extend
* the grammar, but there can be ident in many different places, in
* expression but also in declaration, in function name. So having 3 tokens
* for an ident does not work well with how we add info in
* ast_cpp.ml. So it's better to generate just one token, just one info,
* even if have later to reanalyse those tokens and unsplit.
*
* less: do as in yacfe, generate multiple tokens for those constructs?
*)
| ((id as s) "...")
{ TDefParamVariadic (s, tokinfo lexbuf) }
(* cppext: string concatenation *)
| id ([' ''\t']* "##" [' ''\t']* id)+
{ let info = tokinfo lexbuf in
TIdent (tok lexbuf, info)
}
(* cppext: stringification
* bugfix: this case must be after the other cases such as #endif
* otherwise take precedent.
*)
| "#" (*spopt*) id
{ let info = tokinfo lexbuf in
TIdent (tok lexbuf, info)
}
(* cppext: gccext: ##args for variadic macro *)
| "##" [' ''\t']* id
{ let info = tokinfo lexbuf in
TIdent (tok lexbuf, info)
}
(* only in define body normally *)
| "\\" '\n' { TCppEscapedNewline (tokinfo lexbuf) }
(* ---------------------- *)
(* cpp pragmas *)
(* ---------------------- *)
(* bugfix: I want to keep comments so cant do a sp [^'\n']+ '\n'
* http://gcc.gnu.org/onlinedocs/gcc/Pragmas.html
*)
| "#" spopt "pragma" sp [^'\n']* '\n'
| "#" spopt "ident" sp [^'\n']* '\n'
| "#" spopt "line" sp [^'\n']* '\n'
| "#" spopt "error" sp [^'\n']* '\n'
| "#" spopt "warning" sp [^'\n']* '\n'
| "#" spopt "abort" sp [^'\n']* '\n'
{ TCppDirectiveOther (tokinfo lexbuf) }
(* This appears only after calling cpp cpp, as in:
* # 1 "include/linux/module.h" 1
* Because we handle cpp ourselves, why handle it here?
* Why not ... also one could want to use our parser on
* expanded files sometimes.
*)
| "#" sp pent sp '"' [^ '"']* '"' (spopt pent)* spopt '\n'
{ TCppDirectiveOther (tokinfo lexbuf) }
(* ?? *)
| "#" [' ' '\t']* '\n'
{ TCppDirectiveOther (tokinfo lexbuf) }
(* ----------------------------------------------------------------------- *)
(* C symbols *)
(* ----------------------------------------------------------------------- *)
(* stdC:
* ... && -= >= ~ + ; ]
* <<= &= -> >> % , < ^
* >>= *= /= ^= & - = {
* != ++ << |= ( . > |
* %= += <= || ) / ? }
* -- == ! * : [
* recent addition: <: :> <% %>
* only at processing: %: %:%: # ##
*)
| '[' { TOCro(tokinfo lexbuf) } | ']' { TCCro(tokinfo lexbuf) }
| '(' { TOPar(tokinfo lexbuf) } | ')' { TCPar(tokinfo lexbuf) }
| '{' { TOBrace(tokinfo lexbuf) } | '}' { TCBrace(tokinfo lexbuf) }
| '+' { TPlus(tokinfo lexbuf) } | '*' { TMul(tokinfo lexbuf) }
| '-' { TMinus(tokinfo lexbuf) } | '/' { TDiv(tokinfo lexbuf) }
| '%' { TMod(tokinfo lexbuf) }
| "++"{ TInc(tokinfo lexbuf) } | "--"{ TDec(tokinfo lexbuf) }
| "=" { TEq(tokinfo lexbuf) }
| "-=" { TAssign (OpAssign Minus, (tokinfo lexbuf))}
| "+=" { TAssign (OpAssign Plus, (tokinfo lexbuf))}
| "*=" { TAssign (OpAssign Mul, (tokinfo lexbuf))}
| "/=" { TAssign (OpAssign Div, (tokinfo lexbuf))}
| "%=" { TAssign (OpAssign Mod, (tokinfo lexbuf))}
| "&=" { TAssign (OpAssign And, (tokinfo lexbuf))}
| "|=" { TAssign (OpAssign Or, (tokinfo lexbuf)) }
| "^=" { TAssign(OpAssign Xor, (tokinfo lexbuf))}
| "<<=" {TAssign (OpAssign DecLeft, (tokinfo lexbuf)) }
| ">>=" {TAssign (OpAssign DecRight, (tokinfo lexbuf))}
| "==" { TEqEq(tokinfo lexbuf) } | "!=" { TNotEq(tokinfo lexbuf) }
| ">=" { TSupEq(tokinfo lexbuf) } | "<=" { TInfEq(tokinfo lexbuf) }
(* c++ext: transformed in TInf_Template in parsing_hacks_cpp.ml *)
| "<" { TInf(tokinfo lexbuf) } | ">" { TSup(tokinfo lexbuf) }
| "&&" { TAndLog(tokinfo lexbuf) } | "||" { TOrLog(tokinfo lexbuf) }
| ">>" { TShr(tokinfo lexbuf) } | "<<" { TShl(tokinfo lexbuf) }
| "&" { TAnd(tokinfo lexbuf) } | "|" { TOr(tokinfo lexbuf) }
| "^" { TXor(tokinfo lexbuf) }
| "..." { TEllipsis(tokinfo lexbuf) }
| "->" { TPtrOp(tokinfo lexbuf) } | '.' { TDot(tokinfo lexbuf) }
| ',' { TComma(tokinfo lexbuf) }
| ";" { TPtVirg(tokinfo lexbuf) }
| "?" { TWhy(tokinfo lexbuf) } | ":" { TCol(tokinfo lexbuf) }
| "!" { TBang(tokinfo lexbuf) } | "~" { TTilde(tokinfo lexbuf) }
| "<:" { TOCro(tokinfo lexbuf) } | ":>" { TCCro(tokinfo lexbuf) }
| "<%" { TOBrace(tokinfo lexbuf) } | "%>" { TCBrace(tokinfo lexbuf) }
(* c++ext: *)
| "::" { TColCol(tokinfo lexbuf) }
| "->*" { TPtrOpStar(tokinfo lexbuf) } | ".*" { TDotStar(tokinfo lexbuf) }
(* ----------------------------------------------------------------------- *)
(* C keywords and ident *)
(* ----------------------------------------------------------------------- *)
(* StdC: "must handle at least name of length > 509, but can
* truncate to 31 when compare and truncate to 6 and even lowerise
* in the external linkage phase"
*)
| letter (letter | digit) *
{ let info = tokinfo lexbuf in
let s = tok lexbuf in
Common.profile_code "C parsing.lex_ident" (fun () ->
match Common2.optionise (fun () -> Hashtbl.find keyword_table s) with
| Some f -> f info
(* typedef_hack. note: now this is no more useful, cos
* as we use tokens_all, we first parse then all as idents and
* later transform some idents into typedefs. So this job is
* now done in parse_cpp.ml.
*
* old:
* if Lexer_parser.is_typedef s
* then Ident_Typedef (s, info)
* else TIdent (s, info)
*)
| None -> TIdent (s, info)
)
}
(* gccext: apparently gcc allows dollar in variable names. I've found such
* things a few times in Linux and in glibc.
* No need to look in keyword_table here; definitly a TIdent.
*)
| (letter | '$') (letter | digit | '$')*
{
let s = tok lexbuf in
if not !Flag.sgrep_mode
then error ("identifier with dollar: " ^ s);
TIdent (s, tokinfo lexbuf)
}
(* ----------------------------------------------------------------------- *)
(* C constant *)
(* ----------------------------------------------------------------------- *)
| "'"
{ let info = tokinfo lexbuf in
let s = char lexbuf in
TChar ((s, IsChar), (info +> tok_add_s (s ^ "'")))
}
| '"'
{ let info = tokinfo lexbuf in
let s = string lexbuf in
TString ((s, IsChar), (info +> tok_add_s (s ^ "\"")))
}
(* wide character encoding, TODO L'toto' valid ? what is allowed ? *)
| 'L' "'"
{ let info = tokinfo lexbuf in
let s = char lexbuf in
TChar ((s, IsWchar), (info +> tok_add_s (s ^ "'")))
}
| 'L' '"'
{ let info = tokinfo lexbuf in
let s = string lexbuf in
TString ((s, IsWchar), (info +> tok_add_s (s ^ "\"")))
}
(* Take care of the order ? No because lex try the longest match. The
* strange diff between decimal and octal constant semantic is not
* understood too by refman :) refman:11.1.4, and ritchie.
*)
| (( decimal | hexa | octal)
( ['u' 'U']
| ['l' 'L']
| (['l' 'L'] ['u' 'U'])
| (['u' 'U'] ['l' 'L'])
| (['u' 'U'] ['l' 'L'] ['l' 'L'])
| (['l' 'L'] ['l' 'L'])
)?
) as x { TInt (x, tokinfo lexbuf) }
| (real ['f' 'F']) as x { TFloat ((x, CFloat), tokinfo lexbuf) }
| (real ['l' 'L']) as x { TFloat ((x, CLongDouble), tokinfo lexbuf) }
| (real as x) { TFloat ((x, CDouble), tokinfo lexbuf) }
| ['0'] ['0'-'9']+
{ error (error_radix "octal" ^ tok lexbuf);
TUnknown (tokinfo lexbuf)
}
| ("0x" |"0X") ['0'-'9' 'a'-'z' 'A'-'Z']+
{ error (error_radix "hexa" ^ tok lexbuf);
TUnknown (tokinfo lexbuf)
}
(* !put after other rules! otherwise 0xff will be parsed as an ident *)
| ['0'-'9']+ letter (letter | digit) *
{ error ("ZARB integer_string, certainly a macro:" ^ tok lexbuf);
TUnknown (tokinfo lexbuf)
}
(* gccext: http://gcc.gnu.org/onlinedocs/gcc/Binary-constants.html *)
(*
| "0b" ['0'-'1'] { TInt (((tok lexbuf)<!!>(??,??)) +> int_of_stringbits) }
| ['0'-'1']+'b' { TInt (((tok lexbuf)<!!>(0,-2)) +> int_of_stringbits) }
*)
(*------------------------------------------------------------------------ *)
| eof { EOF (tokinfo lexbuf +> PI.rewrap_str "") }
| _ {
error("unrecognised symbol, in token rule:" ^ tok lexbuf);
TUnknown (tokinfo lexbuf)
}
(*****************************************************************************)
(* Rule char *)
(*****************************************************************************)
and char = parse
(* c++ext: or firefoxext: unicode char may take multiple char as in 'MOSS'
| (_ as x) "'" { String.make 1 x }
(* todo?: as for octal, do exception beyond radix exception ? *)
| (("\\" (oct | oct oct | oct oct oct)) as x "'") { x }
(* this rule must be after the one with octal, lex try first longest
* and when \7 we want an octal, not an exn.
*)
| (("\\x" ((hex | hex hex))) as x "'") { x }
| (("\\" (_ as v)) as x "'")
{
(match v with (* Machine specific ? *)
| 'n' -> () | 't' -> () | 'v' -> () | 'b' -> () | 'r' -> ()
| 'f' -> () | 'a' -> ()
| '\\' -> () | '?' -> () | '\'' -> () | '"' -> ()
| 'e' -> () (* linuxext: ? *)
| _ ->
error ("unrecognised symbol in char:"^tok lexbuf);
);
x
}
| _
{ error ("unrecognised symbol in char:"^tok lexbuf);
tok lexbuf
}
*)
(* c++ext: mostly copy paste of string but s/"/'/ " and s/string/char *)
| '\'' { "" }
| (_ as x)
{ Common2.string_of_char x^char lexbuf}
| ("\\" (oct | oct oct | oct oct oct)) as x { x ^ char lexbuf }
| ("\\x" (hex | hex hex)) as x { x ^ char lexbuf }
| ("\\" (_ as v)) as x
{
(match v with (* Machine specific ? *)
| 'n' -> () | 't' -> () | 'v' -> () | 'b' -> () | 'r' -> ()
| 'f' -> () | 'a' -> ()
| '\\' -> () | '?' -> () | '\'' -> () | '"' -> ()
| 'e' -> () (* linuxext: ? *)
(* old: "x" -> 10 gccext ? todo ugly, I put a fake value *)
(* cppext: can have \ for multiline in string too *)
| '\n' -> ()
| _ -> error ("unrecognised symbol in char:"^tok lexbuf);
);
x ^ char lexbuf
}
| eof { error "WEIRD end of file in char"; ""}
(*****************************************************************************)
(* Rule string *)
(*****************************************************************************)
(* less? factorise code with char ? but not same ending token so hard. *)
and string = parse
| '"' { "" }
| (_ as x)
{ Common2.string_of_char x^string lexbuf}
| ("\\" (oct | oct oct | oct oct oct)) as x { x ^ string lexbuf }
| ("\\x" (hex | hex hex)) as x { x ^ string lexbuf }
(* unicode *)
| ("\\u" (hex hex hex hex)) as x { x ^ string lexbuf }
| ("\\U" (hex hex hex hex hex hex hex hex)) as x { x ^ string lexbuf }
| ("\\" (_ as v)) as x
{
(match v with (* Machine specific ? *)
| 'n' -> () | 't' -> () | 'v' -> () | 'b' -> () | 'r' -> ()
| 'f' -> () | 'a' -> ()
| '\\' -> () | '?' -> () | '\'' -> () | '"' -> ()
| 'e' -> () (* linuxext: ? *)
(* old: "x" -> 10 gccext ? todo ugly, I put a fake value *)
(* cppext: can have \ for multiline in string too *)
| '\n' -> ()
| _ -> error ("unrecognised symbol in string:"^tok lexbuf);
);
x ^ string lexbuf
}
| eof { error "WEIRD end of file in string"; ""}
(* Bug if add following code, cos match also the '"' that is needed
* to finish the string, and so go until end of file.
*)
(*
| [^ '\\']+
{ let cs = lexbuf +> tok +> list_of_string +> List.map Char.code in
cs ++ string lexbuf
}
*)
(*****************************************************************************)
(* Rule comment *)
(*****************************************************************************)
(* less: allow only char-'*' ? *)
and comment = parse
| "*/" { tok lexbuf }
(* noteopti: *)
| [^ '*']+ { let s = tok lexbuf in s ^ comment lexbuf }
| [ '*'] { let s = tok lexbuf in s ^ comment lexbuf }
| _
{ let s = tok lexbuf in
error ("unrecognised symbol in comment:"^s);
s ^ comment lexbuf
}
| eof { error "WEIRD end of file in comment"; ""}
(*****************************************************************************)
(* Rule cpp_eat_until_nl *)
(*****************************************************************************)
(* cpp recognize C comments, so when #define xx (yy) /* comment \n ... */
* then he has already erased the /* comment. So:
* - dont eat the start of the comment otherwise afterwards we are in the middle
* of a comment and so we will problably get a parse error somewhere.
* - have to recognize comments in cpp_eat_until_nl.
*
* note: I was using cpp_eat_until_nl for #define before, but now I
* try also to parse define body so cpp_eat_until_nl is used only for the "body"
* of other uninteresting directtives like #ifdef, #else where can have
* stuff on the right on such directive.
*)
and cpp_eat_until_nl = parse
(* bugfix: need to handle comments too *)
| "/*"
{ let s = tok lexbuf in
let s2 = comment lexbuf in
let s3 = cpp_eat_until_nl lexbuf in
s ^ s2 ^ s3
}
| '\\' "\n" { let s = tok lexbuf in s ^ cpp_eat_until_nl lexbuf }
| "\n" { tok lexbuf }
(* noteopti:
* update: need also deal with comments chars now
*)
| [^ '\n' '\\' '/' '*' ]+
{ let s = tok lexbuf in s ^ cpp_eat_until_nl lexbuf }
| eof { error "end of file in cpp_eat_until_nl"; ""}
| _ { let s = tok lexbuf in s ^ cpp_eat_until_nl lexbuf }

View file

@ -0,0 +1,49 @@
(* Yoann Padioleau
*
* Copyright (C) 2010 Facebook
*
* This library is free software; you can redistribute it and/or
* modify it under the terms of the GNU Lesser General Public License
* version 2.1 as published by the Free Software Foundation, with the
* special exception on linking described in file license.txt.
*
* This library is distributed in the hope that it will be useful, but
* WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the file
* license.txt for more details.
*)
open Common
module V = Visitor_cpp
module FT = File_type
(*****************************************************************************)
(* Filemames *)
(*****************************************************************************)
let find_source_files_of_dir_or_files xs =
Common.files_of_dir_or_files_no_vcs_nofilter xs
+> List.filter (fun filename ->
match File_type.file_type_of_file filename with
| FT.PL (FT.C ("l" | "y")) -> false
| FT.PL (FT.C _ | FT.Cplusplus _ ) ->
(* todo: fix syncweb so don't need this! *)
not (FT.is_syncweb_obj_file filename)
| _ -> false
) +> Common.sort
(*****************************************************************************)
(* ii_of_any *)
(*****************************************************************************)
let ii_of_any any =
let globals = ref [] in
let visitor = V.mk_visitor { V.default_visitor with
V.kinfo = (fun (_k,_) i -> Common.push i globals)
}
in
visitor any;
List.rev !globals

View file

@ -0,0 +1,5 @@
val find_source_files_of_dir_or_files:
Common.path list -> Common.filename list
val ii_of_any: Ast_cpp.any -> Parse_info.info list

View file

@ -0,0 +1,341 @@
GPL
GNU GENERAL PUBLIC LICENSE
Version 2, June 1991
Copyright (C) 1989, 1991 Free Software Foundation, Inc.
675 Mass Ave, Cambridge, MA 02139, USA
Everyone is permitted to copy and distribute verbatim copies
of this license document, but changing it is not allowed.
Preamble
The licenses for most software are designed to take away your
freedom to share and change it. By contrast, the GNU General Public
License is intended to guarantee your freedom to share and change free
software--to make sure the software is free for all its users. This
General Public License applies to most of the Free Software
Foundation's software and to any other program whose authors commit to
using it. (Some other Free Software Foundation software is covered by
the GNU Library General Public License instead.) You can apply it to
your programs, too.
When we speak of free software, we are referring to freedom, not
price. Our General Public Licenses are designed to make sure that you
have the freedom to distribute copies of free software (and charge for
this service if you wish), that you receive source code or can get it
if you want it, that you can change the software or use pieces of it
in new free programs; and that you know you can do these things.
To protect your rights, we need to make restrictions that forbid
anyone to deny you these rights or to ask you to surrender the rights.
These restrictions translate to certain responsibilities for you if you
distribute copies of the software, or if you modify it.
For example, if you distribute copies of such a program, whether
gratis or for a fee, you must give the recipients all the rights that
you have. You must make sure that they, too, receive or can get the
source code. And you must show them these terms so they know their
rights.
We protect your rights with two steps: (1) copyright the software, and
(2) offer you this license which gives you legal permission to copy,
distribute and/or modify the software.
Also, for each author's protection and ours, we want to make certain
that everyone understands that there is no warranty for this free
software. If the software is modified by someone else and passed on, we
want its recipients to know that what they have is not the original, so
that any problems introduced by others will not reflect on the original
authors' reputations.
Finally, any free program is threatened constantly by software
patents. We wish to avoid the danger that redistributors of a free
program will individually obtain patent licenses, in effect making the
program proprietary. To prevent this, we have made it clear that any
patent must be licensed for everyone's free use or not licensed at all.
The precise terms and conditions for copying, distribution and
modification follow.
GNU GENERAL PUBLIC LICENSE
TERMS AND CONDITIONS FOR COPYING, DISTRIBUTION AND MODIFICATION
0. This License applies to any program or other work which contains
a notice placed by the copyright holder saying it may be distributed
under the terms of this General Public License. The "Program", below,
refers to any such program or work, and a "work based on the Program"
means either the Program or any derivative work under copyright law:
that is to say, a work containing the Program or a portion of it,
either verbatim or with modifications and/or translated into another
language. (Hereinafter, translation is included without limitation in
the term "modification".) Each licensee is addressed as "you".
Activities other than copying, distribution and modification are not
covered by this License; they are outside its scope. The act of
running the Program is not restricted, and the output from the Program
is covered only if its contents constitute a work based on the
Program (independent of having been made by running the Program).
Whether that is true depends on what the Program does.
1. You may copy and distribute verbatim copies of the Program's
source code as you receive it, in any medium, provided that you
conspicuously and appropriately publish on each copy an appropriate
copyright notice and disclaimer of warranty; keep intact all the
notices that refer to this License and to the absence of any warranty;
and give any other recipients of the Program a copy of this License
along with the Program.
You may charge a fee for the physical act of transferring a copy, and
you may at your option offer warranty protection in exchange for a fee.
2. You may modify your copy or copies of the Program or any portion
of it, thus forming a work based on the Program, and copy and
distribute such modifications or work under the terms of Section 1
above, provided that you also meet all of these conditions:
a) You must cause the modified files to carry prominent notices
stating that you changed the files and the date of any change.
b) You must cause any work that you distribute or publish, that in
whole or in part contains or is derived from the Program or any
part thereof, to be licensed as a whole at no charge to all third
parties under the terms of this License.
c) If the modified program normally reads commands interactively
when run, you must cause it, when started running for such
interactive use in the most ordinary way, to print or display an
announcement including an appropriate copyright notice and a
notice that there is no warranty (or else, saying that you provide
a warranty) and that users may redistribute the program under
these conditions, and telling the user how to view a copy of this
License. (Exception: if the Program itself is interactive but
does not normally print such an announcement, your work based on
the Program is not required to print an announcement.)
These requirements apply to the modified work as a whole. If
identifiable sections of that work are not derived from the Program,
and can be reasonably considered independent and separate works in
themselves, then this License, and its terms, do not apply to those
sections when you distribute them as separate works. But when you
distribute the same sections as part of a whole which is a work based
on the Program, the distribution of the whole must be on the terms of
this License, whose permissions for other licensees extend to the
entire whole, and thus to each and every part regardless of who wrote it.
Thus, it is not the intent of this section to claim rights or contest
your rights to work written entirely by you; rather, the intent is to
exercise the right to control the distribution of derivative or
collective works based on the Program.
In addition, mere aggregation of another work not based on the Program
with the Program (or with a work based on the Program) on a volume of
a storage or distribution medium does not bring the other work under
the scope of this License.
3. You may copy and distribute the Program (or a work based on it,
under Section 2) in object code or executable form under the terms of
Sections 1 and 2 above provided that you also do one of the following:
a) Accompany it with the complete corresponding machine-readable
source code, which must be distributed under the terms of Sections
1 and 2 above on a medium customarily used for software interchange; or,
b) Accompany it with a written offer, valid for at least three
years, to give any third party, for a charge no more than your
cost of physically performing source distribution, a complete
machine-readable copy of the corresponding source code, to be
distributed under the terms of Sections 1 and 2 above on a medium
customarily used for software interchange; or,
c) Accompany it with the information you received as to the offer
to distribute corresponding source code. (This alternative is
allowed only for noncommercial distribution and only if you
received the program in object code or executable form with such
an offer, in accord with Subsection b above.)
The source code for a work means the preferred form of the work for
making modifications to it. For an executable work, complete source
code means all the source code for all modules it contains, plus any
associated interface definition files, plus the scripts used to
control compilation and installation of the executable. However, as a
special exception, the source code distributed need not include
anything that is normally distributed (in either source or binary
form) with the major components (compiler, kernel, and so on) of the
operating system on which the executable runs, unless that component
itself accompanies the executable.
If distribution of executable or object code is made by offering
access to copy from a designated place, then offering equivalent
access to copy the source code from the same place counts as
distribution of the source code, even though third parties are not
compelled to copy the source along with the object code.
4. You may not copy, modify, sublicense, or distribute the Program
except as expressly provided under this License. Any attempt
otherwise to copy, modify, sublicense or distribute the Program is
void, and will automatically terminate your rights under this License.
However, parties who have received copies, or rights, from you under
this License will not have their licenses terminated so long as such
parties remain in full compliance.
5. You are not required to accept this License, since you have not
signed it. However, nothing else grants you permission to modify or
distribute the Program or its derivative works. These actions are
prohibited by law if you do not accept this License. Therefore, by
modifying or distributing the Program (or any work based on the
Program), you indicate your acceptance of this License to do so, and
all its terms and conditions for copying, distributing or modifying
the Program or works based on it.
6. Each time you redistribute the Program (or any work based on the
Program), the recipient automatically receives a license from the
original licensor to copy, distribute or modify the Program subject to
these terms and conditions. You may not impose any further
restrictions on the recipients' exercise of the rights granted herein.
You are not responsible for enforcing compliance by third parties to
this License.
7. If, as a consequence of a court judgment or allegation of patent
infringement or for any other reason (not limited to patent issues),
conditions are imposed on you (whether by court order, agreement or
otherwise) that contradict the conditions of this License, they do not
excuse you from the conditions of this License. If you cannot
distribute so as to satisfy simultaneously your obligations under this
License and any other pertinent obligations, then as a consequence you
may not distribute the Program at all. For example, if a patent
license would not permit royalty-free redistribution of the Program by
all those who receive copies directly or indirectly through you, then
the only way you could satisfy both it and this License would be to
refrain entirely from distribution of the Program.
If any portion of this section is held invalid or unenforceable under
any particular circumstance, the balance of the section is intended to
apply and the section as a whole is intended to apply in other
circumstances.
It is not the purpose of this section to induce you to infringe any
patents or other property right claims or to contest validity of any
such claims; this section has the sole purpose of protecting the
integrity of the free software distribution system, which is
implemented by public license practices. Many people have made
generous contributions to the wide range of software distributed
through that system in reliance on consistent application of that
system; it is up to the author/donor to decide if he or she is willing
to distribute software through any other system and a licensee cannot
impose that choice.
This section is intended to make thoroughly clear what is believed to
be a consequence of the rest of this License.
8. If the distribution and/or use of the Program is restricted in
certain countries either by patents or by copyrighted interfaces, the
original copyright holder who places the Program under this License
may add an explicit geographical distribution limitation excluding
those countries, so that distribution is permitted only in or among
countries not thus excluded. In such case, this License incorporates
the limitation as if written in the body of this License.
9. The Free Software Foundation may publish revised and/or new versions
of the General Public License from time to time. Such new versions will
be similar in spirit to the present version, but may differ in detail to
address new problems or concerns.
Each version is given a distinguishing version number. If the Program
specifies a version number of this License which applies to it and "any
later version", you have the option of following the terms and conditions
either of that version or of any later version published by the Free
Software Foundation. If the Program does not specify a version number of
this License, you may choose any version ever published by the Free Software
Foundation.
10. If you wish to incorporate parts of the Program into other free
programs whose distribution conditions are different, write to the author
to ask for permission. For software which is copyrighted by the Free
Software Foundation, write to the Free Software Foundation; we sometimes
make exceptions for this. Our decision will be guided by the two goals
of preserving the free status of all derivatives of our free software and
of promoting the sharing and reuse of software generally.
NO WARRANTY
11. BECAUSE THE PROGRAM IS LICENSED FREE OF CHARGE, THERE IS NO WARRANTY
FOR THE PROGRAM, TO THE EXTENT PERMITTED BY APPLICABLE LAW. EXCEPT WHEN
OTHERWISE STATED IN WRITING THE COPYRIGHT HOLDERS AND/OR OTHER PARTIES
PROVIDE THE PROGRAM "AS IS" WITHOUT WARRANTY OF ANY KIND, EITHER EXPRESSED
OR IMPLIED, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF
MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE. THE ENTIRE RISK AS
TO THE QUALITY AND PERFORMANCE OF THE PROGRAM IS WITH YOU. SHOULD THE
PROGRAM PROVE DEFECTIVE, YOU ASSUME THE COST OF ALL NECESSARY SERVICING,
REPAIR OR CORRECTION.
12. IN NO EVENT UNLESS REQUIRED BY APPLICABLE LAW OR AGREED TO IN WRITING
WILL ANY COPYRIGHT HOLDER, OR ANY OTHER PARTY WHO MAY MODIFY AND/OR
REDISTRIBUTE THE PROGRAM AS PERMITTED ABOVE, BE LIABLE TO YOU FOR DAMAGES,
INCLUDING ANY GENERAL, SPECIAL, INCIDENTAL OR CONSEQUENTIAL DAMAGES ARISING
OUT OF THE USE OR INABILITY TO USE THE PROGRAM (INCLUDING BUT NOT LIMITED
TO LOSS OF DATA OR DATA BEING RENDERED INACCURATE OR LOSSES SUSTAINED BY
YOU OR THIRD PARTIES OR A FAILURE OF THE PROGRAM TO OPERATE WITH ANY OTHER
PROGRAMS), EVEN IF SUCH HOLDER OR OTHER PARTY HAS BEEN ADVISED OF THE
POSSIBILITY OF SUCH DAMAGES.
END OF TERMS AND CONDITIONS
How to Apply These Terms to Your New Programs
If you develop a new program, and you want it to be of the greatest
possible use to the public, the best way to achieve this is to make it
free software which everyone can redistribute and change under these terms.
To do so, attach the following notices to the program. It is safest
to attach them to the start of each source file to most effectively
convey the exclusion of warranty; and each file should have at least
the "copyright" line and a pointer to where the full notice is found.
<one line to give the program's name and a brief idea of what it does.>
Copyright (C) 19yy <name of author>
This program is free software; you can redistribute it and/or modify
it under the terms of the GNU General Public License as published by
the Free Software Foundation; either version 2 of the License, or
(at your option) any later version.
This program is distributed in the hope that it will be useful,
but WITHOUT ANY WARRANTY; without even the implied warranty of
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
GNU General Public License for more details.
You should have received a copy of the GNU General Public License
along with this program; if not, write to the Free Software
Foundation, Inc., 675 Mass Ave, Cambridge, MA 02139, USA.
Also add information on how to contact you by electronic and paper mail.
If the program is interactive, make it output a short notice like this
when it starts in an interactive mode:
Gnomovision version 69, Copyright (C) 19yy name of author
Gnomovision comes with ABSOLUTELY NO WARRANTY; for details type `show w'.
This is free software, and you are welcome to redistribute it
under certain conditions; type `show c' for details.
The hypothetical commands `show w' and `show c' should show the appropriate
parts of the General Public License. Of course, the commands you use may
be called something other than `show w' and `show c'; they could even be
mouse-clicks or menu items--whatever suits your program.
You should also get your employer (if you work as a programmer) or your
school, if any, to sign a "copyright disclaimer" for the program, if
necessary. Here is a sample; alter the names:
Yoyodyne, Inc., hereby disclaims all copyright interest in the program
`Gnomovision' (which makes passes at compilers) written by James Hacker.
<signature of Ty Coon>, 1 April 1989
Ty Coon, President of Vice
This General Public License does not permit incorporating your program into
proprietary programs. If your program is a subroutine library, you may
consider it more useful to permit linking proprietary applications with the
library. If this is what you want to do, use the GNU Library General
Public License instead of this License.

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,8 @@
val vof_program:
?precision:Meta_ast_generic.precision ->
Ast_cpp.program -> Ocaml.v
val vof_any:
?precision:Meta_ast_generic.precision ->
Ast_cpp.any -> Ocaml.v

View file

@ -0,0 +1,38 @@
cf engler article about all the difficulty they had because
had to find how to compile code!!
same with elsa, need to know cpp flags.
Well if do certain tools like refactorer, then partial code,
and have not all library code, and lots of other stuff
that cries for a different kind of approach: parse as is.
sgrep_cpp also requires to parse patterns of code passed on the
command line.
related work:
- FrontC of hughes casse
- CIL
- EDG
- semantic designs
- elsa
- cppcheck
- llvm clang
- gcc xml
\section{When pb}
pfff -parse_cpp xxx.cpp
Maybe because of typedef inference is wrong. Had to extend heuristics.
Maybe because template. Had to extend heuristics.
Maybe because of macros. Had to extend macros.h

320
lang_cpp/parsing/orig_c.mly Normal file
View file

@ -0,0 +1,320 @@
%{
(* src: ocamlyaccified from
* http://www.lysator.liu.se/c/ANSI-C-grammar-y.html
*)
open Common
open AbstractSyntax
exception Parsing of string
%}
%token <string * AbstractSyntax.fullType> TString
%token <string> TIdent
%token <int * AbstractSyntax.intType> TInt
%token <float * AbstractSyntax.floatType> TFloat
/*(* conflicts *)*/
%token <string> TypedefIdent
%token TOPar TCPar TOBrace TCBrace TOCro TCCro
%token TDot TComma TPtrOp
%token TInc TDec
%token <AbstractSyntax.assignOp> TAssign
%token TEq
%token TWhy TDotDot TPtVirg TTilde TBang
%token TEllipsis
%token TOrLog TAndLog TOrIncl TOrExcl TAnd TEqEq TNotEq TInf TSup TInfEq TSupEq TShl TShr
TPlus TMinus TMul TDiv TMod
%token Tchar Tshort Tint Tdouble Tfloat Tlong Tunsigned Tsigned Tvoid
Tauto Tregister Textern Tstatic
Tconst Tvolatile
Tstruct Tenum Ttypedef Tunion
Tbreak Telse Tswitch Tcase Tcontinue Tfor Tdo Tif Twhile Treturn Tgoto Tdefault
Tsizeof
%token EOF
%left TOrLog
%left TAndLog
%left TOrIncl
%left TOrExcl
%left TAnd
%left TEqEq TNotEq
%left TInf TSup TInfEq TSupEq
%left TShl TShr
%left TPlus TMinus
%left TMul TDiv TMod
%start main
%type <int list> main
%%
main: translation_unit EOF { [] }
/********************************************************************************/
/*
expression
statement
declaration
main
*/
/********************************************************************************/
expr: assign_expr { }
| expr TComma assign_expr { }
assign_expr: cond_expr { }
| unary_expr TAssign assign_expr { }
| unary_expr TEq assign_expr { }
cond_expr: arith_expr {}
| arith_expr TWhy expr TDotDot cond_expr {}
arith_expr: cast_expr {}
| arith_expr TMul arith_expr {}
| arith_expr TDiv arith_expr {}
| arith_expr TMod arith_expr {}
| arith_expr TPlus arith_expr {}
| arith_expr TMinus arith_expr {}
| arith_expr TShl arith_expr {}
| arith_expr TShr arith_expr {}
| arith_expr TInf arith_expr {}
| arith_expr TSup arith_expr {}
| arith_expr TInfEq arith_expr {}
| arith_expr TSupEq arith_expr {}
| arith_expr TEqEq arith_expr {}
| arith_expr TNotEq arith_expr {}
| arith_expr TAnd arith_expr {}
| arith_expr TOrExcl arith_expr {}
| arith_expr TOrIncl arith_expr {}
| arith_expr TAndLog arith_expr {}
| arith_expr TOrLog arith_expr {}
cast_expr: unary_expr {}
| TOPar type_name TCPar cast_expr {}
unary_expr: postfix_expr {}
| TInc unary_expr {}
| TDec unary_expr {}
| unary_op cast_expr {}
| Tsizeof unary_expr {}
| Tsizeof TOPar type_name TCPar {}
unary_op: TAnd {}
| TMul {}
| TPlus {}
| TMinus{}
| TTilde{}
| TBang {}
postfix_expr: primary_expr {}
| postfix_expr TOCro expr TCCro {}
| postfix_expr TOPar argument_expr_list TCPar {}
| postfix_expr TOPar TCPar {}
| postfix_expr TDot TIdent {}
| postfix_expr TPtrOp TIdent {}
| postfix_expr TInc {}
| postfix_expr TDec {}
argument_expr_list: assign_expr { }
| argument_expr_list TComma assign_expr {}
primary_expr: TIdent {}
| TInt {}
| TFloat {}
| TString {}
| TOPar expr TCPar {}
const_expr: cond_expr {}
/********************************************************************************/
statement: labeled {}
| compound {}
| expr_statement {}
| selection {}
| iteration {}
| jump TPtVirg {}
labeled: TIdent TDotDot statement {}
| Tcase const_expr TDotDot statement {}
| Tdefault TDotDot statement {}
compound: TOBrace TCBrace {}
| TOBrace statement_list TCBrace {}
| TOBrace decl_list TCBrace {}
| TOBrace decl_list statement_list TCBrace {}
decl_list: decl {}
| decl decl_list {}
statement_list: statement {}
| statement statement_list {}
expr_statement: TPtVirg {}
| expr TPtVirg {}
selection: Tif TOPar expr TCPar statement {}
| Tif TOPar expr TCPar statement Telse statement {}
| Tswitch TOPar expr TCPar statement {}
iteration: Twhile TOPar expr TCPar statement {}
| Tdo statement Twhile TOPar expr TCPar TPtVirg {}
| Tfor TOPar expr_statement expr_statement TCPar statement {}
| Tfor TOPar expr_statement expr_statement expr TCPar statement {}
jump: Tgoto TIdent {}
| Tcontinue {}
| Tbreak {}
| Treturn {}
| Treturn expr {}
/********************************************************************************/
/*------------------------------------------------------------------------------*/
decl: decl_spec TPtVirg {}
| decl_spec init_declarator_list TPtVirg {}
/*------------------------------------------------------------------------------*/
decl_spec: storage_class_spec {}
| storage_class_spec decl_spec {}
| type_spec {}
| type_spec decl_spec {}
| type_qualif {}
| type_qualif decl_spec {}
storage_class_spec: Tstatic {}
| Textern {}
| Tauto {}
| Tregister {}
| Ttypedef {}
type_spec: Tvoid {}
| Tchar {}
| Tshort {}
| Tint {}
| Tlong {}
| Tfloat {}
| Tdouble {}
| Tsigned {}
| Tunsigned {}
| struct_or_union_spec {}
| enum_spec {}
/*TODO | TIdent {} */
| TypedefIdent {}
type_qualif: Tconst {}
| Tvolatile {}
/*------------------------------------------------------------------------------*/
struct_or_union_spec: struct_or_union TIdent TOBrace struct_decl_list TCBrace {}
| struct_or_union TOBrace struct_decl_list TCBrace {}
| struct_or_union TIdent {}
struct_or_union: Tstruct {}
| Tunion {}
struct_decl_list: struct_decl {}
| struct_decl_list struct_decl {}
struct_decl: spec_qualif_list struct_declarator_list TPtVirg {}
spec_qualif_list: type_spec {}
| type_spec spec_qualif_list {}
| type_qualif {}
| type_qualif spec_qualif_list {}
struct_declarator_list: struct_declarator {}
| struct_declarator_list TComma struct_declarator {}
struct_declarator: declarator {}
| TDotDot const_expr {}
| declarator TDotDot const_expr {}
/*------------------------------------------------------------------------------*/
enum_spec: Tenum TOBrace enumerator_list TCBrace {}
| Tenum TIdent TOBrace enumerator_list TCBrace {}
| Tenum TIdent {}
enumerator_list: enumerator {}
| enumerator_list TComma enumerator {}
enumerator: TIdent {}
| TIdent TEq const_expr {}
/*------------------------------------------------------------------------------*/
init_declarator_list: init_declarator {}
| init_declarator_list TComma init_declarator {}
init_declarator: declarator {}
| declarator TEq initialize {}
/*------------------------------------------------------------------------------*/
declarator: pointer direct_declarator {}
| direct_declarator {}
pointer: TMul {}
| TMul type_qualif_list {}
| TMul pointer {}
| TMul type_qualif_list pointer {}
direct_declarator: TIdent {}
| TOPar declarator TCPar {}
| direct_declarator TOCro const_expr TCCro {}
| direct_declarator TOCro TCCro {}
| direct_declarator TOPar TCPar {}
| direct_declarator TOPar parameter_type_list TCPar {}
| direct_declarator TOPar identifier_list TCPar {}
type_qualif_list: type_qualif {}
| type_qualif_list type_qualif {}
parameter_type_list: parameter_list {}
| parameter_list TComma TEllipsis {}
parameter_list: parameter_decl {}
| parameter_list TComma parameter_decl {}
parameter_decl: decl_spec declarator {}
| decl_spec abstract_declarator {}
| decl_spec {}
identifier_list: TIdent {}
| identifier_list TComma TIdent {}
/*------------------------------------------------------------------------------*/
type_name: spec_qualif_list {}
| spec_qualif_list abstract_declarator {}
abstract_declarator: pointer {}
| direct_abstract_declarator {}
| pointer direct_abstract_declarator {}
direct_abstract_declarator: TOPar abstract_declarator TCPar {}
| TOCro TCCro {}
| TOCro const_expr TCCro {}
| direct_abstract_declarator TOCro TCCro {}
| direct_abstract_declarator TOCro const_expr TCCro {}
| TOPar TCPar {}
| TOPar parameter_type_list TCPar {}
| direct_abstract_declarator TOPar TCPar {}
| direct_abstract_declarator TOPar parameter_type_list TCPar {}
/*------------------------------------------------------------------------------*/
initialize: assign_expr {}
| TOBrace initialize_list TCBrace {}
| TOBrace initialize_list TComma TCBrace {}
initialize_list: initialize {}
| initialize_list TComma initialize {}
/********************************************************************************/
translation_unit: external_declaration {}
| translation_unit external_declaration {}
external_declaration: function_definition {}
| decl {}
function_definition: decl_spec declarator decl_list compound {}
| decl_spec declarator compound {}
| declarator decl_list compound {}
| declarator compound {}

View file

@ -0,0 +1,759 @@
src: http://www.csci.csusb.edu/dick/c++std/cd2/gram.html
pad: -seq, -opt suffix
1 This summary of C++ syntax is intended to be an aid to comprehension.
It is not an exact statement of the language. In particular, the
grammar described here accepts a superset of valid C++ constructs.
Disambiguation rules (_stmt.ambig_, _dcl.spec_, _class.member.lookup_)
must be applied to distinguish expressions from declarations. Fur-
ther, access control, ambiguity, and type rules must be used to weed
out syntactically valid but meaningless constructs.
1.1 Keywords [gram.key]
1 New context-dependent keywords are introduced into a program by type-
def (_dcl.typedef_), namespace (_namespace.def_), class (_class_),
enumeration (_dcl.enum_), and template (_temp_) declarations.
typedef-name:
identifier
namespace-name:
original-namespace-name
namespace-alias
original-namespace-name:
identifier
namespace-alias:
identifier
class-name:
identifier
template-id
enum-name:
identifier
template-name:
identifier
Note that a typedef-name naming a class is also a class-name
(_class.name_).
1.2 Lexical conventions [gram.lex]
hex-quad:
hexadecimal-digit hexadecimal-digit hexadecimal-digit hexadecimal-digit
universal-character-name:
\u hex-quad
\U hex-quad hex-quad
preprocessing-token:
header-name
identifier
pp-number
character-literal
string-literal
preprocessing-op-or-punc
each non-white-space character that cannot be one of the above
token:
identifier
keyword
literal
operator
punctuator
header-name:
<h-char-sequence>
"q-char-sequence"
h-char-sequence:
h-char
h-char-sequence h-char
h-char:
any member of the source character set except
new-line and >
q-char-sequence:
q-char
q-char-sequence q-char
q-char:
any member of the source character set except
new-line and " "
pp-number:
digit
. digit
pp-number digit
pp-number nondigit
pp-number e sign
pp-number E sign
pp-number .
identifier:
nondigit
identifier nondigit
identifier digit
nondigit: one of
universal-character-name
_ a b c d e f g h i j k l m
n o p q r s t u v w x y z
A B C D E F G H I J K L M
N O P Q R S T U V W X Y Z
digit: one of
0 1 2 3 4 5 6 7 8 9
preprocessing-op-or-punc: one of
{ } [ ] # ## ( )
<: :> <% %> %: %:%: ; : ...
new delete ? :: . .*
+ - * / % ^ & | ~
! = < > += -= *= /= %=
^= &= |= << >> >>= <<= == !=
<= >= && || ++ -- , ->* ->
and and_eq bitand bitor compl not not_eq or or_eq
xor xor_eq
literal:
integer-literal
character-literal
floating-literal
string-literal
boolean-literal
integer-literal:
decimal-literal integer-suffixopt
octal-literal integer-suffixopt
hexadecimal-literal integer-suffixopt
decimal-literal:
nonzero-digit
decimal-literal digit
octal-literal:
0
octal-literal octal-digit
hexadecimal-literal:
0x hexadecimal-digit
0X hexadecimal-digit
hexadecimal-literal hexadecimal-digit
nonzero-digit: one of
1 2 3 4 5 6 7 8 9
octal-digit: one of
0 1 2 3 4 5 6 7
hexadecimal-digit: one of
0 1 2 3 4 5 6 7 8 9
a b c d e f
A B C D E F
integer-suffix:
unsigned-suffix long-suffixopt
long-suffix unsigned-suffixopt
unsigned-suffix: one of
u U
long-suffix: one of
l L
character-literal:
'c-char-sequence'
L'c-char-sequence'
c-char-sequence:
c-char
c-char-sequence c-char
c-char:
any member of the source character set except
the single-quote ', backslash \, or new-line character
escape-sequence
universal-character-name
escape-sequence:
simple-escape-sequence
octal-escape-sequence
hexadecimal-escape-sequence
simple-escape-sequence: one of
\' \" \? \\ "
\a \b \f \n \r \t \v
octal-escape-sequence:
\ octal-digit
\ octal-digit octal-digit
\ octal-digit octal-digit octal-digit
hexadecimal-escape-sequence:
\x hexadecimal-digit
hexadecimal-escape-sequence hexadecimal-digit
floating-literal:
fractional-constant exponent-partopt floating-suffixopt
digit-sequence exponent-part floating-suffixopt
fractional-constant:
digit-sequenceopt . digit-sequence
digit-sequence .
exponent-part:
e signopt digit-sequence
E signopt digit-sequence
sign: one of
+ -
digit-sequence:
digit
digit-sequence digit
floating-suffix: one of
f l F L
string-literal:
"s-char-sequenceopt"
L"s-char-sequenceopt"
s-char-sequence:
s-char
s-char-sequence s-char
s-char:
any member of the source character set except
the double-quote ", backslash \, or new-line character "
escape-sequence
universal-character-name
boolean-literal:
false
true
1.3 Basic concepts [gram.basic]
translation-unit:
declaration-seqopt
1.4 Expressions [gram.expr]
primary-expression:
literal
this
:: identifier
:: operator-function-id
:: qualified-id
( expression )
id-expression
id-expression:
unqualified-id
qualified-id
unqualified-id:
identifier
operator-function-id
conversion-function-id
~ class-name
template-id
qualified-id:
nested-name-specifier templateopt unqualified-id
nested-name-specifier:
class-or-namespace-name :: nested-name-specifieropt
class-or-namespace-name:
class-name
namespace-name
postfix-expression:
primary-expression
postfix-expression [ expression ]
postfix-expression ( expression-listopt )
simple-type-specifier ( expression-listopt )
postfix-expression . templateopt ::opt id-expression
postfix-expression -> templateopt ::opt id-expression
postfix-expression . pseudo-destructor-name
postfix-expression -> pseudo-destructor-name
postfix-expression ++
postfix-expression --
dynamic_cast < type-id > ( expression )
static_cast < type-id > ( expression )
reinterpret_cast < type-id > ( expression )
const_cast < type-id > ( expression )
typeid ( expression )
typeid ( type-id )
expression-list:
assignment-expression
expression-list , assignment-expression
pseudo-destructor-name:
::opt nested-name-specifieropt type-name :: ~ type-name
::opt nested-name-specifieropt ~ type-name
unary-expression:
postfix-expression
++ cast-expression
-- cast-expression
unary-operator cast-expression
sizeof unary-expression
sizeof ( type-id )
new-expression
delete-expression
unary-operator: one of
* & + - ! ~
new-expression:
::opt new new-placementopt new-type-id new-initializeropt
::opt new new-placementopt ( type-id ) new-initializeropt
new-placement:
( expression-list )
new-type-id:
type-specifier-seq new-declaratoropt
new-declarator:
ptr-operator new-declaratoropt
direct-new-declarator
direct-new-declarator:
[ expression ]
direct-new-declarator [ constant-expression ]
new-initializer:
( expression-listopt )
delete-expression:
::opt delete cast-expression
::opt delete [ ] cast-expression
cast-expression:
unary-expression
( type-id ) cast-expression
pm-expression:
cast-expression
pm-expression .* cast-expression
pm-expression ->* cast-expression
multiplicative-expression:
pm-expression
multiplicative-expression * pm-expression
multiplicative-expression / pm-expression
multiplicative-expression % pm-expression
additive-expression:
multiplicative-expression
additive-expression + multiplicative-expression
additive-expression - multiplicative-expression
shift-expression:
additive-expression
shift-expression << additive-expression
shift-expression >> additive-expression
relational-expression:
shift-expression
relational-expression < shift-expression
relational-expression > shift-expression
relational-expression <= shift-expression
relational-expression >= shift-expression
equality-expression:
relational-expression
equality-expression == relational-expression
equality-expression != relational-expression
and-expression:
equality-expression
and-expression & equality-expression
exclusive-or-expression:
and-expression
exclusive-or-expression ^ and-expression
inclusive-or-expression:
exclusive-or-expression
inclusive-or-expression | exclusive-or-expression
logical-and-expression:
inclusive-or-expression
logical-and-expression && inclusive-or-expression
logical-or-expression:
logical-and-expression
logical-or-expression || logical-and-expression
conditional-expression:
logical-or-expression
logical-or-expression ? expression : assignment-expression
assignment-expression:
conditional-expression
logical-or-expression assignment-operator assignment-expression
throw-expression
assignment-operator: one of
= *= /= %= += -= >>= <<= &= ^= |=
expression:
assignment-expression
expression , assignment-expression
constant-expression:
conditional-expression
1.5 Statements [gram.stmt.stmt]
statement:
labeled-statement
expression-statement
compound-statement
selection-statement
iteration-statement
jump-statement
declaration-statement
try-block
labeled-statement:
identifier : statement
case constant-expression : statement
default : statement
expression-statement:
expressionopt ;
compound-statement:
{ statement-seqopt }
statement-seq:
statement
statement-seq statement
selection-statement:
if ( condition ) statement
if ( condition ) statement else statement
switch ( condition ) statement
condition:
expression
type-specifier-seq declarator = assignment-expression
iteration-statement:
while ( condition ) statement
do statement while ( expression ) ;
for ( for-init-statement conditionopt ; expressionopt ) statement
for-init-statement:
expression-statement
simple-declaration
jump-statement:
break ;
continue ;
return expressionopt ;
goto identifier ;
declaration-statement:
block-declaration
1.6 Declarations [gram.dcl.dcl]
declaration-seq:
declaration
declaration-seq declaration
declaration:
block-declaration
function-definition
template-declaration
explicit-instantiation
explicit-specialization
linkage-specification
namespace-definition
block-declaration:
simple-declaration
asm-definition
namespace-alias-definition
using-declaration
using-directive
simple-declaration:
decl-specifier-seqopt init-declarator-listopt ;
decl-specifier:
storage-class-specifier
type-specifier
function-specifier
friend
typedef
decl-specifier-seq:
decl-specifier-seqopt decl-specifier
storage-class-specifier:
auto
register
static
extern
mutable
function-specifier:
inline
virtual
explicit
typedef-name:
identifier
type-specifier:
simple-type-specifier
class-specifier
enum-specifier
elaborated-type-specifier
cv-qualifier
simple-type-specifier:
::opt nested-name-specifieropt type-name
char
wchar_t
bool
short
int
long
signed
unsigned
float
double
void
type-name:
class-name
enum-name
typedef-name
elaborated-type-specifier:
class-key ::opt nested-name-specifieropt identifier
enum ::opt nested-name-specifieropt identifier
typename ::opt nested-name-specifier identifier
typename ::opt nested-name-specifier identifier < template-argument-list >
enum-name:
identifier
enum-specifier:
enum identifieropt { enumerator-listopt }
enumerator-list:
enumerator-definition
enumerator-list , enumerator-definition
enumerator-definition:
enumerator
enumerator = constant-expression
enumerator:
identifier
namespace-name:
original-namespace-name
namespace-alias
original-namespace-name:
identifier
namespace-definition:
named-namespace-definition
unnamed-namespace-definition
named-namespace-definition:
original-namespace-definition
extension-namespace-definition
original-namespace-definition:
namespace identifier { namespace-body }
extension-namespace-definition:
namespace original-namespace-name { namespace-body }
unnamed-namespace-definition:
namespace { namespace-body }
namespace-body:
declaration-seqopt
namespace-alias:
identifier
namespace-alias-definition:
namespace identifier = qualified-namespace-specifier ;
qualified-namespace-specifier:
::opt nested-name-specifieropt namespace-name
using-declaration:
using typenameopt ::opt nested-name-specifier unqualified-id ;
using :: unqualified-id ;
using-directive:
using namespace ::opt nested-name-specifieropt namespace-name ;
asm-definition:
asm ( string-literal ) ;
linkage-specification:
extern string-literal { declaration-seqopt }
extern string-literal declaration
1.7 Declarators [gram.dcl.decl]
init-declarator-list:
init-declarator
init-declarator-list , init-declarator
init-declarator:
declarator initializeropt
declarator:
direct-declarator
ptr-operator declarator
direct-declarator:
declarator-id
direct-declarator ( parameter-declaration-clause ) cv-qualifier-seqopt exception-specificationopt
direct-declarator [ constant-expressionopt ]
( declarator )
ptr-operator:
* cv-qualifier-seqopt
&
::opt nested-name-specifier * cv-qualifier-seqopt
cv-qualifier-seq:
cv-qualifier cv-qualifier-seqopt
cv-qualifier:
const
volatile
declarator-id:
::opt id-expression
::opt nested-name-specifieropt type-name
type-id:
type-specifier-seq abstract-declaratoropt
type-specifier-seq:
type-specifier type-specifier-seqopt
abstract-declarator:
ptr-operator abstract-declaratoropt
direct-abstract-declarator
direct-abstract-declarator:
direct-abstract-declaratoropt ( parameter-declaration-clause ) cv-qualifier-seqopt exception-specificationopt
direct-abstract-declaratoropt [ constant-expressionopt ]
( abstract-declarator )
parameter-declaration-clause:
parameter-declaration-listopt ...opt
parameter-declaration-list , ...
parameter-declaration-list:
parameter-declaration
parameter-declaration-list , parameter-declaration
parameter-declaration:
decl-specifier-seq declarator
decl-specifier-seq declarator = assignment-expression
decl-specifier-seq abstract-declaratoropt
decl-specifier-seq abstract-declaratoropt = assignment-expression
function-definition:
decl-specifier-seqopt declarator ctor-initializeropt function-body
decl-specifier-seqopt declarator function-try-block
function-body:
compound-statement
initializer:
= initializer-clause
( expression-list )
initializer-clause:
assignment-expression
{ initializer-list ,opt }
{ }
initializer-list:
initializer-clause
initializer-list , initializer-clause
1.8 Classes [gram.class]
class-name:
identifier
template-id
class-specifier:
class-head { member-specificationopt }
class-head:
class-key identifieropt base-clauseopt
class-key nested-name-specifier identifier base-clauseopt
class-key:
class
struct
union
member-specification:
member-declaration member-specificationopt
access-specifier : member-specificationopt
member-declaration:
decl-specifier-seqopt member-declarator-listopt ;
function-definition ;opt
qualified-id ;
using-declaration
template-declaration
member-declarator-list:
member-declarator
member-declarator-list , member-declarator
member-declarator:
declarator pure-specifieropt
declarator constant-initializeropt
identifieropt : constant-expression
pure-specifier:
= 0
constant-initializer:
= constant-expression
1.9 Derived classes [gram.class.derived]
base-clause:
: base-specifier-list
base-specifier-list:
base-specifier
base-specifier-list , base-specifier
base-specifier:
::opt nested-name-specifieropt class-name
virtual access-specifieropt ::opt nested-name-specifieropt class-name
access-specifier virtualopt ::opt nested-name-specifieropt class-name
access-specifier:
private
protected
public
1.10 Special member functions [gram.special]
conversion-function-id:
operator conversion-type-id
conversion-type-id:
type-specifier-seq conversion-declaratoropt
conversion-declarator:
ptr-operator conversion-declaratoropt
ctor-initializer:
: mem-initializer-list
mem-initializer-list:
mem-initializer
mem-initializer , mem-initializer-list
mem-initializer:
mem-initializer-id ( expression-listopt )
mem-initializer-id:
::opt nested-name-specifieropt class-name
identifier
1.11 Overloading [gram.over]
operator-function-id:
operator operator
operator: one of
new delete new[] delete[]
+ - * / % ^ & | ~
! = < > += -= *= /= %=
^= &= |= << >> >>= <<= == !=
<= >= && || ++ -- , ->* ->
() []
1.12 Templates [gram.temp]
template-declaration:
exportopt template < template-parameter-list > declaration
template-parameter-list:
template-parameter
template-parameter-list , template-parameter
template-parameter:
type-parameter
parameter-declaration
type-parameter:
class identifieropt
class identifieropt = type-id
typename identifieropt
typename identifieropt = type-id
template < template-parameter-list > class identifieropt
template < template-parameter-list > class identifieropt = template-name
template-id:
template-name < template-argument-list >
template-name:
identifier
template-argument-list:
template-argument
template-argument-list , template-argument
template-argument:
assignment-expression
type-id
template-name
explicit-instantiation:
template-declaration
explicit-specialization:
template < > declaration
1.13 Exception handling [gram.except]
try-block:
try compound-statement handler-seq
function-try-block:
try ctor-initializeropt function-body handler-seq
handler-seq:
handler handler-seqopt
handler:
catch ( exception-declaration ) compound-statement
exception-declaration:
type-specifier-seq declarator
type-specifier-seq abstract-declarator
type-specifier-seq
...
throw-expression:
throw assignment-expressionopt
exception-specification:
throw ( type-id-listopt )
type-id-list:
type-id
type-id-list , type-id
1.14 Preprocessing directives [gram.cpp]
preprocessing-file:
groupopt
group:
group-part
group group-part
group-part:
pp-tokensopt new-line
if-section
control-line
if-section:
if-group elif-groupsopt else-groupopt endif-line
if-group:
# if constant-expression new-line groupopt
# ifdef identifier new-line groupopt
# ifndef identifier new-line groupopt
elif-groups:
elif-group
elif-groups elif-group
elif-group:
# elif constant-expression new-line groupopt
else-group:
# else new-line groupopt
endif-line:
# endif new-line
control-line:
# include pp-tokens new-line
# define identifier replacement-list new-line
# define identifier lparen identifier-listopt ) replacement-list new-line
# undef identifier new-line
# line pp-tokens new-line
# error pp-tokensopt new-line
# pragma pp-tokensopt new-line
# new-line
lparen:
the left-parenthesis character without preceding white-space
replacement-list:
pp-tokensopt
pp-tokens:
preprocessing-token
pp-tokens preprocessing-token
new-line:
the new-line character

View file

@ -0,0 +1,506 @@
(* Yoann Padioleau
*
* Copyright (C) 2002-2013 Yoann Padioleau
*
* This program is free software; you can redistribute it and/or
* modify it under the terms of the GNU General Public License (GPL)
* version 2 as published by the Free Software Foundation.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* file license.txt for more details.
*)
open Common
module Ast = Ast_cpp
module Flag = Flag_parsing_cpp
module PI = Parse_info
module Stat = Parse_info
module T = Parser_cpp
module TH = Token_helpers_cpp
module Lexer = Lexer_cpp
module Semantic = Parser_cpp_mly_helper
module Hack = Parsing_hacks_lib
module FT = File_type
(*****************************************************************************)
(* Prelude *)
(*****************************************************************************)
(*
* A heuristic based C/cpp/C++ parser.
*
* See "Parsing C/C++ Code without Pre-Preprocessing - Yoann Padioleau, CC'09"
* avalaible at http://padator.org/papers/yacfe-cc09.pdf
*)
(*****************************************************************************)
(* Types *)
(*****************************************************************************)
type toplevels_and_tokens = (Ast.toplevel * Parser_cpp.token list) list
let program_of_program2 xs =
xs +> List.map fst
exception Parse_error of Parse_info.info
(*****************************************************************************)
(* Wrappers *)
(*****************************************************************************)
let pr2, _pr2_once = Common2.mk_pr2_wrappers Flag_parsing_cpp.verbose_parsing
(*****************************************************************************)
(* Error diagnostic *)
(*****************************************************************************)
let error_msg_tok tok =
Parse_info.error_message_info (TH.info_of_tok tok)
(*****************************************************************************)
(* Stats on what was passed/commentized *)
(*****************************************************************************)
let commentized xs = xs +> Common.map_filter (function
| T.TComment_Pp (cppkind, ii) ->
if !Flag.filter_classic_passed
then
(match cppkind with
| Token_cpp.CppOther ->
let s = PI.str_of_info ii in
(match s with
| s when s =~ "KERN_.*" -> None
| s when s =~ "__.*" -> None
| _ -> Some (ii.PI.token)
)
| Token_cpp.CppDirective | Token_cpp.CppAttr | Token_cpp.CppMacro
-> None
| Token_cpp.CppMacroExpanded
| Token_cpp.CppPassingNormal
| Token_cpp.CppPassingCosWouldGetError
-> raise Todo
)
else Some (ii.PI.token)
| T.TAny_Action ii ->
Some (ii.PI.token)
| _ ->
None
)
let count_lines_commentized xs =
let line = ref (-1) in
let count = ref 0 in
commentized xs +> List.iter (function
| PI.OriginTok pinfo
| PI.ExpandedTok (_,pinfo,_) ->
let newline = pinfo.PI.line in
if newline <> !line
then begin
line := newline;
incr count
end
| _ -> ()
);
!count
(* See also problematic_lines and parsing_stat.ml *)
(* for most problematic tokens *)
let is_same_line_or_close line tok =
TH.line_of_tok tok =|= line ||
TH.line_of_tok tok =|= line - 1 ||
TH.line_of_tok tok =|= line - 2
(*****************************************************************************)
(* Lexing only *)
(*****************************************************************************)
(* called by parse below *)
let tokens2 file =
let table = Parse_info.full_charpos_to_pos_large file in
Common.with_open_infile file (fun chan ->
let lexbuf = Lexing.from_channel chan in
try
let rec tokens_aux () =
let tok = Lexer.token lexbuf in
(* fill in the line and col information *)
let tok = tok +> TH.visitor_info_of_tok (fun ii ->
{ ii with PI.token=
(* could assert pinfo.filename = file ? *)
match ii.PI.token with
| PI.OriginTok pi ->
PI.OriginTok (Parse_info.complete_token_location_large file
table pi)
| PI.ExpandedTok (pi,vpi, off) ->
PI.ExpandedTok(
(Parse_info.complete_token_location_large file table pi),vpi,
off)
| PI.FakeTokStr (s,vpi_opt) -> PI.FakeTokStr (s,vpi_opt)
| PI.Ab -> raise Impossible
})
in
if TH.is_eof tok
then [tok]
else tok::(tokens_aux ())
in
tokens_aux ()
with
| Lexer.Lexical s ->
failwith (spf "lexical error %s \n = %s"
s (PI.error_message file (PI.lexbuf_to_strpos lexbuf)))
| e -> raise e
)
let tokens a =
Common.profile_code "Parse_cpp.tokens" (fun () -> tokens2 a)
(*****************************************************************************)
(* Fuzzy parsing *)
(*****************************************************************************)
let rec multi_grouped_list xs =
xs +> List.map multi_grouped
and multi_grouped = function
| Token_views_cpp.Braces (tok1, xs, (Some tok2)) ->
Ast_fuzzy.Braces (tokext tok1, multi_grouped_list xs, tokext tok2)
| Token_views_cpp.Parens (tok1, xs, (Some tok2)) ->
Ast_fuzzy.Parens (tokext tok1, multi_grouped_list_comma xs, tokext tok2)
| Token_views_cpp.Angle (tok1, xs, (Some tok2)) ->
Ast_fuzzy.Angle (tokext tok1, multi_grouped_list xs, tokext tok2)
| Token_views_cpp.Tok (tok) ->
(match PI.str_of_info (tokext tok) with
| "..." -> Ast_fuzzy.Dots (tokext tok)
| s when Ast_fuzzy.is_metavar s -> Ast_fuzzy.Metavar (s, tokext tok)
| s -> Ast_fuzzy.Tok (s, tokext tok)
)
| _ -> failwith "could not find closing brace/parens/angle"
and tokext tok_extended =
TH.info_of_tok tok_extended.Token_views_cpp.t
and multi_grouped_list_comma xs =
let rec aux acc xs =
match xs with
| [] ->
if null acc
then []
else [Left (acc +> List.rev +> multi_grouped_list)]
| (x::xs) ->
(match x with
| Token_views_cpp.Tok tok when PI.str_of_info (tokext tok) = "," ->
let before = acc +> List.rev +> multi_grouped_list in
if null before
then aux [] xs
else (Left before)::(Right (tokext tok))::aux [] xs
| _ ->
aux (x::acc) xs
)
in
aux [] xs
(* This is similar to what I did for OPA. This is also similar
* to what I do for parsing hacks, but this fuzzy AST can be useful
* on its own, e.g. for a not too bad sgrep/spatch.
*
* note: this is similar to what cpplint/fblint of andrei does?
*)
let parse_fuzzy file =
Common.save_excursion Flag.sgrep_mode true (fun () ->
let toks_orig = tokens file in
let toks =
toks_orig +> Common.exclude (fun x ->
Token_helpers_cpp.is_comment x || Token_helpers_cpp.is_eof x
)
in
let extended = toks +> List.map Token_views_cpp.mk_token_extended in
Parsing_hacks_cpp.find_template_inf_sup extended;
let groups = Token_views_cpp.mk_multi extended in
multi_grouped_list groups, toks_orig
)
(*****************************************************************************)
(* Extract macros *)
(*****************************************************************************)
(* It can be used to to parse the macros defined in a macro.h file. It
* can also be used to try to extract the macros defined in the file
* that we try to parse *)
let extract_macros2 file =
Common.save_excursion Flag_parsing_cpp.verbose_lexing false (fun () ->
let toks = tokens (* todo: ~profile:false *) file in
let toks = Parsing_hacks_define.fix_tokens_define toks in
Pp_token.extract_macros toks
)
let extract_macros a =
Common.profile_code_exclusif "Parse_cpp.extract_macros" (fun () ->
extract_macros2 a)
(* less: pass it as a parameter to parse_program instead ?
* old: was a ref, but a hashtbl.t is actually already a kind of ref
*)
let (_defs : (string, Pp_token.define_body) Hashtbl.t) =
Hashtbl.create 101
(* We used to have also a init_defs_builtins() so that we could use a
* standard.h containing macros that were always useful, and a macros.h
* that the user could customize for his own project.
* But this was adding complexity so now we just have _defs and people
* can call add_defs to add local macro definitions.
*)
let add_defs file =
if not (Sys.file_exists file)
then failwith (spf "Could not find %s, have you set PFFF_HOME correctly?"
file);
pr2 (spf "Using %s macro file" file);
let xs = extract_macros file in
xs +> List.iter (fun (k, v) -> Hashtbl.add _defs k v)
let init_defs file =
Hashtbl.clear _defs;
add_defs file
(*****************************************************************************)
(* Error recovery *)
(*****************************************************************************)
(* see parsing_recovery_cpp.ml *)
(*****************************************************************************)
(* Consistency checking *)
(*****************************************************************************)
(* todo: a parsing_consistency_cpp.ml *)
(*****************************************************************************)
(* Helper for main entry point *)
(*****************************************************************************)
(* Hacked lex. This function use refs passed by parse.
* 'tr' means 'token refs'. This is used mostly to enable
* error recovery (This used to do lots of stuff, such as
* calling some lookahead heuristics to reclassify
* tokens such as TIdent into TIdent_Typeded but this is
* now done in a fix_tokens style in parsing_hacks_typedef.ml.
*)
let rec lexer_function tr = fun lexbuf ->
match tr.PI.rest with
| [] -> (pr2 "LEXER: ALREADY AT END"; tr.PI.current)
| v::xs ->
tr.PI.rest <- xs;
tr.PI.current <- v;
tr.PI.passed <- v::tr.PI.passed;
if !Flag.debug_lexer then pr2_gen v;
if TH.is_comment v
then lexer_function (*~pass*) tr lexbuf
else v
(* was a define ? *)
let passed_a_define tr =
let xs = tr.PI.passed +> List.rev +> Common.exclude TH.is_comment in
if List.length xs >= 2
then
(match Common2.head_middle_tail xs with
| T.TDefine _, _, T.TCommentNewline_DefineEndOfMacro _ -> true
| _ -> false
)
else begin
pr2 "WIERD: length list of error recovery tokens < 2 ";
false
end
(*****************************************************************************)
(* Main entry point *)
(*****************************************************************************)
(*
* note: as now we go in two passes, there is first all the error message of
* the lexer, and then the error of the parser. It is not anymore
* interwinded.
*
* !!!This function use refs, and is not reentrant !!! so take care.
* It uses the _defs global defined above!!!!
*)
let parse_with_lang ?(lang=Flag_parsing_cpp.Cplusplus) file =
let stat = Parse_info.default_stat file in
let filelines = Common2.cat_array file in
(* -------------------------------------------------- *)
(* call lexer and get all the tokens *)
(* -------------------------------------------------- *)
let toks_orig = tokens file in
let toks =
try Parsing_hacks.fix_tokens ~macro_defs:_defs lang toks_orig
with Token_views_cpp.UnclosedSymbol s ->
pr2 s;
if !Flag.debug_cplusplus
then raise (Token_views_cpp.UnclosedSymbol s)
else toks_orig
in
let tr = Parse_info.mk_tokens_state toks in
let lexbuf_fake = Lexing.from_function (fun _buf _n -> raise Impossible) in
let rec loop () =
let info = TH.info_of_tok tr.PI.current in
(* todo?: I am not sure that it represents current_line, cos maybe
* tr.current partipated in the previous parsing phase, so maybe tr.current
* is not the first token of the next parsing phase. Same with checkpoint2.
* It would be better to record when we have a } or ; in parser.mly,
* cos we know that they are the last symbols of external_declaration2.
*)
let checkpoint = PI.line_of_info info in
(* bugfix: may not be equal to 'file' as after macro expansions we can
* start to parse a new entity from the body of a macro, for instance
* when parsing a define_machine() body, cf standard.h
*)
let checkpoint_file = PI.file_of_info info in
tr.PI.passed <- [];
(* for some statistics *)
let was_define = ref false in
let elem =
(try
(* -------------------------------------------------- *)
(* Call parser *)
(* -------------------------------------------------- *)
Parser_cpp.toplevel (lexer_function tr) lexbuf_fake
with e ->
if not !Flag.error_recovery
then raise (Parse_error (TH.info_of_tok tr.PI.current));
if !Flag.show_parsing_error then
(match e with
(* Lexical is not anymore launched I think *)
| Lexer.Lexical s ->
pr2 ("lexical error " ^s^ "\n =" ^ error_msg_tok tr.PI.current)
| Parsing.Parse_error ->
pr2 ("parse error \n = " ^ error_msg_tok tr.PI.current)
| Semantic.Semantic (s, _i) ->
pr2 ("semantic error " ^s^ "\n ="^ error_msg_tok tr.PI.current)
| e -> raise e
);
let line_error = TH.line_of_tok tr.PI.current in
let pbline =
tr.PI.passed
+> List.filter (is_same_line_or_close line_error)
+> List.filter TH.is_ident_like
in
let error_info =
(pbline +> List.map (fun tok->PI.str_of_info (TH.info_of_tok tok))),
line_error
in
stat.Stat.problematic_lines <-
error_info::stat.Stat.problematic_lines;
(* error recovery, go to next synchro point *)
let (passed', rest') =
Parsing_recovery_cpp.find_next_synchro tr.PI.rest tr.PI.passed in
tr.PI.rest <- rest';
tr.PI.passed <- passed';
tr.PI.current <- List.hd passed';
(* <> line_error *)
let info = TH.info_of_tok tr.PI.current in
let checkpoint2 = PI.line_of_info info in
let checkpoint2_file = PI.file_of_info info in
was_define := passed_a_define tr;
(if !was_define && !Flag.filter_define_error
then ()
else
(* bugfix: *)
(if (checkpoint_file = checkpoint2_file) && checkpoint_file = file
then PI.print_bad line_error (checkpoint, checkpoint2) filelines
else pr2 "PB: bad: but on tokens not from original file"
)
);
let info_of_bads =
Common2.map_eff_rev TH.info_of_tok tr.PI.passed in
Some (Ast.NotParsedCorrectly info_of_bads)
)
in
(* again not sure if checkpoint2 corresponds to end of bad region *)
let info = TH.info_of_tok tr.PI.current in
let checkpoint2 = PI.line_of_info info in
let checkpoint2_file = PI.file_of_info info in
let diffline =
if (checkpoint_file = checkpoint2_file) && (checkpoint_file = file)
then (checkpoint2 - checkpoint)
else 0
(* TODO? so if error come in middle of something ? where the
* start token was from original file but synchro found in body
* of macro ? then can have wrong number of lines stat.
* Maybe simpler just to look at tr.passed and count
* the lines in the token from the correct file ?
*)
in
let info = List.rev tr.PI.passed in
(* some stat updates *)
stat.Stat.commentized <-
stat.Stat.commentized + count_lines_commentized info;
(match elem with
| Some (Ast.NotParsedCorrectly _xs) ->
if !was_define && !Flag.filter_define_error
then stat.Stat.commentized <- stat.Stat.commentized + diffline
else stat.Stat.bad <- stat.Stat.bad + diffline
| _ -> stat.Stat.correct <- stat.Stat.correct + diffline
);
(match elem with
| None -> []
| Some xs -> (xs, info):: loop () (* recurse *)
)
in
let v = loop() in
(v, stat)
let parse2 file =
match File_type.file_type_of_file file with
| FT.PL (FT.C _) ->
(try
parse_with_lang ~lang:Flag.C file
with _exn ->
parse_with_lang ~lang:Flag.Cplusplus file
)
| FT.PL (FT.Cplusplus _) ->
parse_with_lang ~lang:Flag.Cplusplus file
| _ -> failwith (spf "not a C/C++ file: %s" file)
let parse file =
Common.profile_code "Parse_cpp.parse" (fun () ->
try
parse2 file
with Stack_overflow ->
pr2 (spf "PB stack overflow in %s" file);
[(Ast.NotParsedCorrectly [], ([]))], {Stat.
correct = 0;
bad = Common2.nblines_with_wc file;
filename = file;
have_timeout = true;
commentized = 0;
problematic_lines = [];
}
)
let parse_program file =
let (ast2, _stat) = parse file in
program_of_program2 ast2

View file

@ -0,0 +1,43 @@
(* the token list contains also the comment-tokens *)
type toplevels_and_tokens = (Ast_cpp.toplevel * Parser_cpp.token list) list
(* actually covers Lexical, Parsing, and Semantic errors *)
exception Parse_error of Parse_info.info
(* This is the main function. It uses _defs below which often comes
* from a standard.h macro file. It will raise Parse_error unless
* Flag_parsing_cpp.error_recovery is set.
*)
val parse:
Common.filename -> (toplevels_and_tokens * Parse_info.parsing_stat)
val parse_program:
Common.filename -> Ast_cpp.program
val parse_with_lang:
?lang:Flag_parsing_cpp.language ->
Common.filename -> (toplevels_and_tokens * Parse_info.parsing_stat)
val parse_fuzzy:
Common.filename -> Ast_fuzzy.tree list * Parser_cpp.token list
(* usually correspond to what is inside your macros.h *)
val _defs : (string, Pp_token.define_body) Hashtbl.t
val init_defs : Common.filename -> unit
val add_defs : Common.filename -> unit
(* used to extract macros from standard.h, but also now used on C files
* in -extract_macros to assist in building a macros.h
*)
val extract_macros:
Common.filename -> (string, Pp_token.define_body) Common.assoc
(* usually correspond to what is inside your standard.h *)
(* val _defs_builtins : (string, Cpp_token_c.define_def) Hashtbl.t ref *)
(* todo: init_defs_macros and init_defs_builtins *)
(* subsystem testing *)
val tokens: Common.filename -> Parser_cpp.token list
(* a few helpers *)
val program_of_program2: toplevels_and_tokens -> Ast_cpp.program

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,344 @@
open Common
open Ast_cpp
module Ast = Ast_cpp
module Flag = Flag_parsing_cpp
(*****************************************************************************)
(* Wrappers *)
(*****************************************************************************)
let pr2, pr2_once = Common2.mk_pr2_wrappers Flag.verbose_parsing
let warning s v =
if !Flag.verbose_parsing
then Common2.warning ("PARSING: " ^ s) v
else v
exception Semantic of string * Ast_cpp.tok
(*****************************************************************************)
(* Parse helpers functions *)
(*****************************************************************************)
(*-------------------------------------------------------------------------- *)
(* Type related *)
(*-------------------------------------------------------------------------- *)
type shortLong = Short | Long | LongLong
(* note: have a full_info: parse_info list; to remember ordering
* between storage, qualifier, type? well this info is already in
* the Ast_c.info, just have to sort them to get good order
*)
type decl = {
storageD: storage;
typeD: (sign option * shortLong option * typeCbis option) wrap;
qualifD: typeQualifier;
inlineD: bool wrap;
}
let nullDecl = {
storageD = NoSto;
typeD = (None, None, None), noii;
qualifD = Ast.nQ;
inlineD = false, noii;
}
let addStorageD x decl =
match decl with
| {storageD = NoSto; _} -> { decl with storageD = x }
| {storageD = (StoTypedef ii | Sto (_, ii)) as y; _} ->
if x = y
then decl +> warning "duplicate storage classes"
else raise (Semantic ("multiple storage classes", ii))
let addInlineD ii decl =
match decl with
| {inlineD = (false,[]); _} -> { decl with inlineD=(true,[ii])}
| {inlineD = (true, _ii2); _} -> decl +> warning "duplicate inline"
| _ -> raise Impossible
let addTypeD ty decl =
match ty, decl with
| (Left3 Signed,_ii), {typeD = ((Some Signed, _b,_c),_ii2); _} ->
decl +> warning "duplicate 'signed'"
| (Left3 UnSigned,_ii), {typeD = ((Some UnSigned,_b,_c),_ii2); _} ->
decl +> warning "duplicate 'unsigned'"
| (Left3 _,ii), {typeD = ((Some _,_b,_c),_ii2); _} ->
raise (Semantic ("both signed and unsigned specified", List.hd ii))
| (Left3 x,ii), {typeD = ((None,b,c),ii2); _} ->
{ decl with typeD = (Some x,b,c),ii @ ii2}
| (Middle3 Short,_ii), {typeD = ((_a,Some Short,_c),_ii2); _} ->
decl +> warning "duplicate 'short'"
(* gccext: long long allowed *)
| (Middle3 Long,ii), {typeD = ((a,Some Long,c),ii2); _}->
{ decl with typeD = (a, Some LongLong, c),ii@ii2 }
| (Middle3 Long,_ii), {typeD = ((_a,Some LongLong,_c),_ii2); _} ->
decl +> warning "triplicate 'long'"
| (Middle3 _,ii), {typeD = ((_a,Some _,_c),_ii2); _} ->
raise (Semantic ("both long and short specified", List.hd ii))
| (Middle3 x,ii), {typeD = ((a,None,c),ii2); _} ->
{ decl with typeD = (a, Some x,c),ii@ii2}
| (Right3 _t,ii), {typeD = ((_a,_b,Some _),_ii2); _} ->
raise (Semantic ("two or more data types", List.hd ii))
| (Right3 t,ii), {typeD = ((a,b,None),ii2); _} ->
{ decl with typeD = (a,b, Some t),ii@ii2}
let addQualif tq1 tq2 =
match tq1, tq2 with
| {const=Some _; _}, {const=Some _; _} ->
tq2 +> warning "duplicate 'const'"
| {volatile=Some _; _}, {volatile=Some _; _} ->
tq2 +> warning "duplicate 'volatile'"
| {const=Some x; _}, _ ->
{ tq2 with const = Some x}
| {volatile=Some x; _}, _ ->
{ tq2 with volatile = Some x}
| _ -> Common2.internal_error "there is no noconst or novolatile keyword"
let addQualifD qu qu2 =
{ qu2 with qualifD = addQualif qu qu2.qualifD }
(*-------------------------------------------------------------------------- *)
(* Declaration/Function related *)
(*-------------------------------------------------------------------------- *)
(* stdC: type section, basic integer types (and ritchie)
* To understand the code, just look at the result (right part of the PM)
* and go back.
*)
let type_and_storage_from_decl
{storageD = st;
qualifD = qu;
typeD = (ty,iit);
inlineD = (inline,iinl);
} =
(qu,
(match ty with
| (None, None, None) ->
(* mine (originally default to int, but this looks like bad style) *)
let decl =
{ v_namei = None; v_type = qu, (BaseType Void, iit); v_storage = st } in
raise (Semantic ("no type (could default to 'int')",
List.hd (Lib_parsing_cpp.ii_of_any (OneDecl decl))))
| (None, None, Some t) -> (t, iit)
| (Some sign, None, (None| Some (BaseType (IntType (Si (_,CInt)))))) ->
BaseType(IntType (Si (sign, CInt))), iit
| ((None|Some Signed),Some x,(None|Some(BaseType(IntType (Si (_,CInt)))))) ->
BaseType(IntType (Si (Signed, [Short,CShort; Long, CLong; LongLong, CLongLong] +> List.assoc x))), iit
| (Some UnSigned, Some x, (None| Some (BaseType (IntType (Si (_,CInt))))))->
BaseType(IntType (Si (UnSigned, [Short,CShort; Long, CLong; LongLong, CLongLong] +> List.assoc x))), iit
| (Some sign, None, (Some (BaseType (IntType CChar)))) -> BaseType(IntType (Si (sign, CChar2))), iit
| (None, Some Long,(Some(BaseType(FloatType CDouble)))) -> BaseType (FloatType (CLongDouble)), iit
| (Some _,_, Some _) ->
raise (Semantic("signed, unsigned valid only for char and int", List.hd iit))
| (_,Some _,(Some(BaseType(FloatType (CFloat|CLongDouble))))) ->
raise (Semantic ("long or short specified with floatint type", List.hd iit))
| (_,Some Short,(Some(BaseType(FloatType CDouble)))) ->
raise (Semantic ("the only valid combination is long double", List.hd iit))
| (_, Some _, Some _) ->
(* mine *)
raise (Semantic ("long, short valid only for int or float", List.hd iit))
(* if do short uint i, then gcc say parse error, strange ? it is
* not a parse error, it is just that we dont allow with typedef
* either short/long or signed/unsigned. In fact, with
* parse_typedef_fix2 (with et() and dt()) now I say too parse
* error so this code is executed only when do short struct
* {....} and never with a typedef cos now we parse short uint i
* as short ident ident => parse error (cos after first short i
* pass in dt() mode) *)
)), st, (inline, iinl)
let type_and_register_from_decl decl =
let {storageD = st; _} = decl in
let (t,_storage, _inline) = type_and_storage_from_decl decl in
match st with
| NoSto -> t, None
| Sto (Register, ii) -> t, Some ii
| StoTypedef ii | Sto (_, ii) ->
raise (Semantic ("storage class specified for parameter of function", ii))
let fixNameForParam (name, ftyp) =
match name with
| None, [], IdIdent id -> id, ftyp
| _ ->
let ii = Lib_parsing_cpp.ii_of_any (Name name) +> List.hd in
raise (Semantic ("parameter have qualifier", ii))
let type_and_storage_for_funcdef_from_decl decl =
let (returnType, storage, _inline) = type_and_storage_from_decl decl in
(match storage with
| StoTypedef tok ->
raise (Semantic ("function definition declared 'typedef'", tok))
| _x -> (returnType, storage)
)
(*
* this function is used for func definitions (not declarations).
* In that case we must have a name for the parameter.
* This function ensures that we give only parameterTypeDecl with well
* formed Classic constructor.
*
* todo?: do we accept other declaration in ?
* so I must add them to the compound of the deffunc. I dont
* have to handle typedef pb here cos C forbid to do VF f { ... }
* with VF a typedef of func cos here we dont see the name of the
* argument (in the typedef)
*)
let (fixOldCDecl: fullType -> fullType) = fun ty ->
match snd ty with
| FunctionType ({ft_params=params;_}),_iifunc ->
(* stdC: If the prototype declaration declares a parameter for a
* function that you are defining (it is part of a function
* definition), then you must write a name within the declarator.
* Otherwise, you can omit the name. *)
(match Ast.unparen params with
| [{p_name = None; p_type = ty2;_},_] ->
(match Ast.unwrap_typeC ty2 with
| BaseType Void -> ty
| _ ->
(* less: there is some valid case actually, when use interfaces
* and generic callbacks where specific instances do not
* need the extra parameter (happens a lot in plan9).
* Maybe this check is better done in a scheck for C.
let info = Lib_parsing_cpp.ii_of_any (Type ty2) +> List.hd in
pr2 (spf "SEMANTIC: parameter name omitted (but I continue) at %s"
(Parse_info.string_of_info info)
);
*)
ty
)
| params ->
(params +> List.iter (fun (param,_) ->
match param with
| {p_name = None; p_type = _ty2; _} ->
(* see above
let info = Lib_parsing_cpp.ii_of_any (Type ty2) +> List.hd in
(* if majuscule, then certainly macro-parameter *)
pr2 (spf "SEMANTIC: parameter name omitted (but I continue) at %s"
(Parse_info.string_of_info info)
);
*)
()
| _ -> ()
));
ty
)
(* todo? can we declare prototype in the decl or structdef,
* ... => length <> but good kan meme
*)
| _ ->
(* gcc says parse error but I dont see why *)
let ii = Lib_parsing_cpp.ii_of_any (Type ty) +> List.hd in
raise (Semantic ("seems this is not a function", ii))
(* TODO: this is ugly ... use record! *)
let fixFunc ((name, ty, sto), cp) =
match ty with
| (aQ,(FunctionType ({ft_params=params; _} as ftyp),_iifunc)) ->
(* it must be nullQualif, cos parser construct only this *)
assert (aQ =*= nQ);
(match Ast.unparen params with
[{p_name= None; p_type = ty2;_}, _] ->
(match Ast.unwrap_typeC ty2 with
| BaseType Void -> ()
(* failwith "internal errror: fixOldCDecl not good" *)
| _ -> ()
)
| params ->
params +> List.iter (function
| ({p_name = Some _s;_}, _) -> ()
(* failwith "internal errror: fixOldCDecl not good" *)
| _ -> ()
)
);
{ f_name = name; f_type = ftyp; f_storage = sto; f_body = cp; }
| _ ->
let ii = Lib_parsing_cpp.ii_of_any (Type ty) +> List.hd in
raise (Semantic ("function definition without parameters", ii))
let fixFieldOrMethodDecl (xs, semicolon) =
match xs with
| [FieldDecl({
v_namei = Some (name, ini_opt);
v_type = (q, (FunctionType ft, ii_ft));
v_storage = sto;
}), _noiicomma] ->
(* todo? define another type instead of onedecl? *)
MemberDecl (MethodDecl ({
v_namei = Some (name, None);
v_type = (q, (FunctionType ft, ii_ft));
v_storage = sto;
},
(match ini_opt with
| None -> None
| Some (EqInit(tokeq, InitExpr(C(Int "0"), iizero))) ->
Some (tokeq, List.hd iizero)
| _ ->
raise (Semantic ("can't assign expression to method decl", semicolon))
), semicolon
))
| _ -> MemberField (xs, semicolon)
(*-------------------------------------------------------------------------- *)
(* shortcuts *)
(*-------------------------------------------------------------------------- *)
let mk_e e ii = (e, ii)
let mk_funcall e1 args =
Call (e1, args)
let mk_constructor id (lp, params, rp) cp =
let params, _hasdots =
match params with
| Some (params, ellipsis) ->
params, ellipsis
| None -> [], None
in
let ftyp = {
ft_ret = nQ, (BaseType Void, noii);
ft_params= (lp, params, rp);
ft_dots = None;
(* TODO *)
ft_const = None;
ft_throw = None;
}
in
{ f_name = (None, noQscope, IdIdent id); f_type = ftyp;
f_storage = NoSto; f_body = cp
}
let mk_destructor tilde id (lp, _voidopt, rp) exnopt cp =
let ftyp = {
ft_ret = nQ, (BaseType Void, noii);
ft_params= (lp, [], rp);
ft_dots = None;
ft_const = None;
ft_throw = exnopt;
}
in
{ f_name = (None, noQscope, IdDestructor (tilde, id)); f_type = ftyp;
f_storage = NoSto; f_body = cp;
}
let opt_to_list_params params =
match params with
| Some (params, _ellipsis) ->
(* todo? raise a warning that should not have ellipsis? *)
params
| None -> []

View file

@ -0,0 +1,277 @@
(* Yoann Padioleau
*
* Copyright (C) 2011,2014 Facebook
* Copyright (C) 2002-2008 Yoann Padioleau
*
* This program is free software; you can redistribute it and/or
* modify it under the terms of the GNU General Public License (GPL)
* version 2 as published by the Free Software Foundation.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* file license.txt for more details.
*)
open Common
module Flag = Flag_parsing_cpp
module TH = Token_helpers_cpp
module TV = Token_views_cpp
module T = Parser_cpp
module PI = Parse_info
(*****************************************************************************)
(* Prelude *)
(*****************************************************************************)
(*
* This module tries to detect some cpp, C, or C++ idioms so that we can
* parse as-is files by adjusting or commenting some tokens.
*
* Sometimes we use some name conventions, sometimes indentation information,
* sometimes we do some kind of lalr(k) by finding patterns. We often try to
* work on a better token representation, like ifdef-paren-ized, brace-ized,
* paren-ized, so that we can pattern-match more easily
* complex idioms (see token_views_cpp.ml).
* We also try to get more contextual information such as whether the
* token is in an initializer because many idioms are different
* depending on the context (see token_views_context.ml).
*
* Examples of cpp idioms:
* - if 0 for commenting stuff (not always code, sometimes any text)
* - ifdef old version
* - ifdef funheader
* - ifdef statements, ifdef expression, ifdef-mid
* - macro toplevel (with or without a trailing ';')
* - macro foreach
* - macro higher order
* - macro declare
* - macro debug
* - macro no ';'
* - macro string, and macro function string taking param and ##
* - macro attribute
*
* Examples of C typedef idioms:
* - x * y
*
* Examples of C++ idioms:
* - x<...> for templates. People rarely do x < y > z to express
* relational expressions, so a < followed later by a > is probably a
* template.
*
* See the TIdent_MacroXxx in parser_cpp.mly and MacroXxx in ast_cpp.ml
*
* We also do other stuff involving cpp like expanding macros,
* and we try to parse define body by finding the end of define virtual
* end-of-line token. But now most of the code is actually in pp_token.ml
* It is related to what is in the yacfe configuration file (e.g. standard.h)
*)
(*****************************************************************************)
(* Helpers *)
(*****************************************************************************)
let filter_comment_stuff xs =
xs +> List.filter (fun x -> not (TH.is_comment x.TV.t))
(*****************************************************************************)
(* Post processing *)
(*****************************************************************************)
(* to do at the very very end *)
let insert_virtual_positions l =
let strlen x = String.length (Parse_info.str_of_info x) in
let rec loop prev offset = function
[] -> []
| x::xs ->
let ii = TH.info_of_tok x in
let inject pi =
TH.visitor_info_of_tok (function ii -> Ast_cpp.rewrap_pinfo pi ii)x in
match ii.Parse_info.token with
Parse_info.OriginTok _pi ->
let prev = Parse_info.token_location_of_info ii in
x::(loop prev (strlen ii) xs)
| Parse_info.ExpandedTok (pi,_, _) ->
inject (Parse_info.ExpandedTok (pi, prev,offset)) ::
(loop prev (offset + (strlen ii)) xs)
| Parse_info.FakeTokStr (s,_) ->
inject (Parse_info.FakeTokStr (s, (Some (prev,offset)))) ::
(loop prev (offset + (strlen ii)) xs)
| Parse_info.Ab -> failwith "abstract not expected" in
let rec skip_fake = function
[] -> []
| x::xs ->
let ii = TH.info_of_tok x in
match ii.Parse_info.token with
Parse_info.OriginTok _pi ->
let prev = Parse_info.token_location_of_info ii in
x::(loop prev (strlen ii) xs)
| _ -> x::skip_fake xs in
skip_fake l
(*****************************************************************************)
(* C vs C++ *)
(*****************************************************************************)
let fix_tokens_for_language lang xs =
xs +> List.map (fun tok ->
if lang = Flag_parsing_cpp.C && TH.is_cpp_keyword tok
then
let ii = TH.info_of_tok tok in
T.TIdent (PI.str_of_info ii, ii)
else tok
)
(*****************************************************************************)
(* Fix tokens *)
(*****************************************************************************)
(*
* Main entry point for the token reclassifier which generates "fresh" tokens.
*
* The order of the rules is important. For instance if you put the
* action heuristic first, then because of ifdef, can have not closed paren
* and so may believe that higher order macro
* and it will eat too much tokens. So important to do
* first the ifdef heuristic.
*
* Note that the functions below work on a list of token_extended
* or on views on top of a list of token_extended. The token_extended record
* contains mutable fields which explains the (ugly but working) imperative
* style of the code below.
*
* I recompute multiple times 'cleaner' cos the mutable
* can have be changed and so we may have more comments
* in the token original list.
*)
(* we could factorize with fix_tokens_cpp, but for debugging purpose it
* might be good to have two different functions and do far less in
* fix_tokens_c (even though the extra steps in fix_tokens_cpp should
* have no effect on regular C code).
*)
let fix_tokens_c ~macro_defs tokens =
let tokens = Parsing_hacks_define.fix_tokens_define tokens in
let tokens = fix_tokens_for_language Flag.C tokens in
let tokens2 = ref (tokens +> Common2.acc_map TV.mk_token_extended) in
(* ifdef *)
let cleaner = !tokens2 +> filter_comment_stuff in
let ifdef_grouped = TV.mk_ifdef cleaner in
Parsing_hacks_pp.find_ifdef_funheaders ifdef_grouped;
Parsing_hacks_pp.find_ifdef_bool ifdef_grouped;
Parsing_hacks_pp.find_ifdef_mid ifdef_grouped;
(* macro part 1 *)
let cleaner = !tokens2 +> Parsing_hacks_pp.filter_pp_or_comment_stuff in
let paren_grouped = TV.mk_parenthised cleaner in
Pp_token.apply_macro_defs macro_defs paren_grouped;
(* because the before field is used by apply_macro_defs *)
tokens2 := TV.rebuild_tokens_extented !tokens2;
let cleaner = !tokens2 +> Parsing_hacks_pp.filter_pp_or_comment_stuff in
let paren_grouped = TV.mk_parenthised cleaner in
Parsing_hacks_pp.find_define_init_brace_paren paren_grouped;
Parsing_hacks_pp.find_string_macro_paren paren_grouped;
Parsing_hacks_pp.find_macro_paren paren_grouped;
let cleaner = !tokens2 +> Parsing_hacks_pp.filter_pp_or_comment_stuff in
(* tagging contextual info (InFunc, InStruct, etc) *)
let multi_grouped = TV.mk_multi cleaner in
Token_views_context.set_context_tag_multi multi_grouped;
let xxs = Parsing_hacks_typedef.filter_for_typedef multi_grouped in
Parsing_hacks_typedef.find_typedefs xxs;
insert_virtual_positions (!tokens2 +> Common2.acc_map (fun x -> x.TV.t))
let fix_tokens_cpp ~macro_defs tokens =
let tokens = Parsing_hacks_define.fix_tokens_define tokens in
(* let tokens = fix_tokens_for_language Flag.Cplusplus tokens in *)
let tokens2 = ref (tokens +> Common2.acc_map TV.mk_token_extended) in
(* ifdef *)
let cleaner = !tokens2 +> filter_comment_stuff in
let ifdef_grouped = TV.mk_ifdef cleaner in
Parsing_hacks_pp.find_ifdef_funheaders ifdef_grouped;
Parsing_hacks_pp.find_ifdef_bool ifdef_grouped;
Parsing_hacks_pp.find_ifdef_mid ifdef_grouped;
(* macro part 1 *)
let cleaner = !tokens2 +> Parsing_hacks_pp.filter_pp_or_comment_stuff in
(* find '<' '>' template symbols. We need that for the typedef
* heuristics. We actually need that even for the paren view
* which is wrong without it.
*
* todo? expand macro first? some expand to lexical_cast ...
* but need correct parenthized view to expand macros => mutually recursive :(
*)
Parsing_hacks_cpp.find_template_inf_sup cleaner;
let paren_grouped = TV.mk_parenthised cleaner in
Pp_token.apply_macro_defs macro_defs paren_grouped;
(* because the before field is used by apply_macro_defs *)
tokens2 := TV.rebuild_tokens_extented !tokens2;
(* could filter also #define/#include *)
let cleaner = !tokens2 +> filter_comment_stuff in
(* tagging contextual info (InFunc, InStruct, etc). Better to do
* that after the "ifdef-simplification" phase.
*)
let multi_grouped = TV.mk_multi cleaner in
Token_views_context.set_context_tag_multi multi_grouped;
(* macro part 2 *)
let cleaner = !tokens2 +> Parsing_hacks_pp.filter_pp_or_comment_stuff in
let paren_grouped = TV.mk_parenthised cleaner in
let line_paren_grouped = TV.mk_line_parenthised paren_grouped in
Parsing_hacks_pp.find_define_init_brace_paren paren_grouped;
Parsing_hacks_pp.find_string_macro_paren paren_grouped;
Parsing_hacks_pp.find_macro_lineparen line_paren_grouped;
Parsing_hacks_pp.find_macro_paren paren_grouped;
(* todo: at some point we need to remove that and use
* a better filter_for_typedef that also
* works on the nested template arguments.
*)
Parsing_hacks_cpp.find_template_commentize multi_grouped;
let cleaner = !tokens2 +> Parsing_hacks_pp.filter_pp_or_comment_stuff in
(* must be done before the qualifier filtering *)
Parsing_hacks_cpp.find_constructor_outside_class cleaner;
Parsing_hacks_cpp.find_qualifier_commentize cleaner;
let cleaner = !tokens2 +> Parsing_hacks_pp.filter_pp_or_comment_stuff in
let multi_grouped = TV.mk_multi cleaner in
Token_views_context.set_context_tag_cplus multi_grouped;
Parsing_hacks_cpp.find_constructor cleaner;
let xxs = Parsing_hacks_typedef.filter_for_typedef multi_grouped in
Parsing_hacks_typedef.find_typedefs xxs;
(* must be done after the typedef inference *)
Parsing_hacks_cpp.find_constructed_object_and_more cleaner;
(* the pending of find_qualifier_comentize *)
Parsing_hacks_cpp.reclassify_tokens_before_idents_or_typedefs multi_grouped;
insert_virtual_positions (!tokens2 +> Common2.acc_map (fun x -> x.TV.t))
let fix_tokens ~macro_defs lang a =
Common.profile_code "C++ parsing.fix_tokens" (fun () ->
match lang with
| Flag_parsing_cpp.C -> fix_tokens_c ~macro_defs a
| Flag_parsing_cpp.Cplusplus -> fix_tokens_cpp ~macro_defs a
)

View file

@ -0,0 +1,7 @@
(* will among other things interally call pp_token.ml to expand some macros *)
val fix_tokens:
macro_defs:(string, Pp_token.define_body) Hashtbl.t ->
Flag_parsing_cpp.language ->
Parser_cpp.token list -> Parser_cpp.token list

View file

@ -0,0 +1,476 @@
(* Yoann Padioleau
*
* Copyright (C) 2002-2008 Yoann Padioleau
* Copyright (C) 2011 Facebook
*
* This program is free software; you can redistribute it and/or
* modify it under the terms of the GNU General Public License (GPL)
* version 2 as published by the Free Software Foundation.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* file license.txt for more details.
*)
open Common
module Flag = Flag_parsing_cpp
module Ast = Ast_cpp
module TH = Token_helpers_cpp
module TV = Token_views_cpp
module Parser = Parser_cpp
module PI = Parse_info
open Parser_cpp
open Token_views_cpp
open Parsing_hacks_lib
(*****************************************************************************)
(* Prelude *)
(*****************************************************************************)
(*
* This file gathers parsing heuristics related to C++.
* See also Token_views_cpp.set_context_tag and
* Parsing_hacks_typedef.filter_for_typedef that have
* heuristics specific to C++.
*
* TODO: * TIdent_TemplatenameInQualifier
*
*)
(*****************************************************************************)
(* Helpers *)
(*****************************************************************************)
let no_space_between i1 i2 =
(PI.line_of_info i1 = PI.line_of_info i2) &&
(PI.col_of_info i1 + String.length (PI.str_of_info i1))= PI.col_of_info i2
(*****************************************************************************)
(* Template inference *)
(*****************************************************************************)
let templateLOOKAHEAD = 30
(* note: no need to check for TCPar to stop for instance the search,
* this is will be done automatically because we would be inside a
* Parenthised expression.
*)
let rec have_a_tsup_quite_close xs =
match xs with
| [] -> false
| x::xs ->
(match x with
| {t=TSup _} -> true
(* false positive *)
| {t=tok} when TH.is_static_cast_like tok -> false
(* ugly: *)
| {t=(TOBrace _ | TPtVirg _ | TCol _ | TAssign _ )} ->
false
| {t=TInf _} ->
(* probably nested template, still try
* TODO: bug when have i < DEG<...>::foo(...)
* we should recurse!
*)
have_a_tsup_quite_close xs
(* bugfix: but want allow some binary operator :) like '*' *)
| {t=tok} when TH.is_binary_operator_except_star tok -> false
| _ -> have_a_tsup_quite_close xs
)
(* precondition: there is a tsup *)
let rec find_tsup_quite_close tok_open xs =
let rec aux acc xs =
match xs with
| [] ->
raise (UnclosedSymbol
(spf "PB: find_tsup_quite_close, no > for < at line %d"
(TH.line_of_tok tok_open.t)))
| x::xs ->
(match x with
| {t=TSup ii} ->
List.rev acc, (x,ii), xs
| {t=TInf _} ->
(* recurse *)
let (before, (tsuptok,_), after) = find_tsup_quite_close x xs in
(* we don't care about this one, it will be eventually be
* transformed by the caller *)
aux (tsuptok:: (List.rev before) @(x::acc)) after
| x -> aux (x::acc) xs
)
in
aux [] xs
(* note: some macros in standard.h may expand to static_cast, so perhaps
* better to do template detection after macro expansion ?
*
* C-s for TInf_Template in the grammar and you will see all cases
* should be covered by the patterns below.
*)
let find_template_inf_sup xs =
let rec aux xs =
match xs with
| [] -> ()
(* template<...> *)
| {t=Ttemplate _}::({t=TInf i2} as tok2)::xs ->
change_tok tok2 (TInf_Template i2);
let (before_sup, (toksup, toksupi), rest) =
find_tsup_quite_close tok2 xs in
change_tok toksup (TSup_Template toksupi);
(* recurse *)
aux before_sup;
aux rest
(* static_cast<...> *)
| {t=tok1}::({t=TInf i2} as tok2)::xs
when TH.is_static_cast_like tok1 ->
change_tok tok2 (TInf_Template i2);
let (before_sup, (toksup, toksupi), rest) =
find_tsup_quite_close tok2 xs in
change_tok toksup (TSup_Template toksupi);
(* recurse *)
aux before_sup;
aux rest
(*
* TODO: have_a_tsup_quite_close does not handle a relational < followed
* by a regular template.
*)
| {t=TIdent (_,i1)}::({t=TInf i2} as tok2)::xs
when
no_space_between i1 i2 && (* safe guard, and good style anyway *)
have_a_tsup_quite_close (Common.take_safe templateLOOKAHEAD xs)
->
change_tok tok2 (TInf_Template i2);
let (before_sup, (toksup, toksupi), rest) =
find_tsup_quite_close tok2 xs in
change_tok toksup (TSup_Template toksupi);
(* old: was changing to TIdent_Templatename but now first need
* to do the typedef inference and then can transform the
* TIdent_Typedef into a TIdent_Templatename
*)
(* recurse *)
aux before_sup;
aux rest
(* special cases which allow extra space between ident and <
* but I think it would be better for people to fix their code
* | {t=TIdent (s,i1)}::({t=TInf i2} as tok2)
* ::tok3::({t=TSup i4} as tok4)::xs ->
* ...
*
*)
(* recurse *)
| _::xs -> aux xs
in
aux xs
(*****************************************************************************)
(* Main heuristics *)
(*****************************************************************************)
let reclassify_tokens_before_idents_or_typedefs xs =
let groups = List.rev xs in
let rec aux xs =
match xs with
| [] -> ()
(* xx::yy where yy is ident (funcall, variable, etc)
* need to do that recursively! if have a::b::c
*)
| Tok{t=TIdent _ | TIdent_ClassnameInQualifier _}
::Tok{t=TColCol _}
::Tok({t=TIdent (s2, i2)} as tok2)::xs ->
change_tok tok2 (TIdent_ClassnameInQualifier (s2, i2));
aux ((Tok tok2)::xs)
(* xx::t wher et is a type
* TODO need to do that recursively! if have a::b::c
*)
| Tok{t=TIdent_Typedef _}::Tok({t=TColCol icolcol} as tcolcol)
::Tok({t=TIdent (s2, i2)} as tok2)::xs ->
change_tok tok2 (TIdent_ClassnameInQualifier_BeforeTypedef (s2, i2));
change_tok tcolcol (TColCol_BeforeTypedef icolcol);
aux xs
(* xx::t<...> where t is a templatename *)
| Tok{t=TIdent_Templatename _}::Tok({t=TColCol icolcol} as tcolcol)
::Tok({t=TIdent (s2, i2)} as tok2)::xs ->
change_tok tok2 (TIdent_ClassnameInQualifier_BeforeTypedef (s2, i2));
change_tok tcolcol (TColCol_BeforeTypedef icolcol);
aux xs
(* t<...> where t is a typedef *)
| Angle (_, xs_angle, _)::Tok({t=TIdent_Typedef (s1, i1)} as tok1)::xs ->
aux xs_angle;
change_tok tok1 (TIdent_Templatename (s1, i1));
(* recurse with tok1 too! *)
aux (Tok tok1::xs)
(* TODO
* TIdent_TemplatenameInQualifier ?
*)
| x::xs ->
(match x with
| Tok _ -> ()
| Braces (_, xs, _)
| Parens (_, xs, _)
| Angle (_, xs, _)
-> aux (List.rev xs)
);
aux xs
in
aux groups;
()
(* quite similar to filter_for_typedef
* TODO: at some point need have to remove this and instead
* have a correct filter_for_typedef that also returns
* nested types in template arguments (and some
* typedef heuristics that work on template_arguments too)
*
* TODO: once you don't use it, remove certain grammar rules (C-s TODO)
*)
let find_template_commentize groups =
(* remove template *)
let rec aux xs =
xs +> List.iter (function
| TV.Braces (_, xs, _) ->
aux xs
| TV.Parens (_, xs, _) ->
aux xs
| TV.Angle (_, _xs, _) as angle ->
(* let's commentize everything *)
[angle] +> TV.iter_token_multi (fun tok ->
change_tok tok
(TComment_Cpp (Token_cpp.CplusplusTemplate, TH.info_of_tok tok.t))
)
| TV.Tok tok ->
(* todo? should also pass the static_cast<...> which normally
* expect some TInf_Template after. Right mow I manage
* that by having some extra rules in the grammar
*)
(match tok.t with
| Ttemplate _ ->
change_tok tok
(TComment_Cpp (Token_cpp.CplusplusTemplate, TH.info_of_tok tok.t))
| _ -> ()
)
)
in
aux groups
(* assumes a view without:
* - template arguments
*
* TODO: once you don't use it, remove certain grammar rules (C-s TODO)
*
* note: passing qualifiers is slightly less important than passing template
* arguments because they are before the name (as opposed to templates
* which are after) and most of our heuristics for typedefs
* look tokens forward, not backward (actually a few now look backward too)
*)
let find_qualifier_commentize xs =
let rec aux xs =
match xs with
| [] -> ()
| ({t=TIdent _} as t1)::({t=TColCol _} as t2)::xs ->
[t1; t2] +> List.iter (fun tok ->
change_tok tok
(TComment_Cpp (Token_cpp.CplusplusQualifier, TH.info_of_tok tok.t))
);
aux xs
(* need also to pass the top :: *)
| ({t=TColCol _} as t2)::xs ->
[t2] +> List.iter (fun tok ->
change_tok tok
(TComment_Cpp (Token_cpp.CplusplusQualifier, TH.info_of_tok tok.t))
);
aux xs
(* recurse *)
| _::xs ->
aux xs
in
aux xs
(* assumes a view where:
* - set_context_tag has been called.
* TODO: filter the 'explicit' keyword? filter the TCppDirectiveOther
* have a filter_for_constructor?
*)
let find_constructor xs =
let rec aux xs =
match xs with
| [] -> ()
(* { Foo(... *)
| {t=(TOBrace _ | TCBrace _ | TPtVirg _ | Texplicit _);_}
::({t=TIdent (s1, i1); where=(TV.InClassStruct s2)::_; _} as tok1)
::{t=TOPar _}::xs when s1 = s2 ->
change_tok tok1 (TIdent_Constructor(s1, i1));
aux xs
(* public: Foo(... could also filter the privacy directives so
* need only one rule
*)
| {t=(Tpublic _ | Tprotected _ | Tprivate _)}::{t=TCol _}
::({t=TIdent (s1, i1); where=(TV.InClassStruct s2)::_; _} as tok1)
::{t=TOPar _}::xs when s1 = s2 ->
change_tok tok1 (TIdent_Constructor(s1, i1));
aux xs
(* recurse *)
| _::xs -> aux xs
in
aux xs
(* assumes a view where:
* - template have been filtered but NOT the qualifiers!
*)
let find_constructor_outside_class xs =
let rec aux xs =
match xs with
| [] -> ()
| {t=TIdent (s1, _);_}::{t=TColCol _}::({t=TIdent (s2,i2);_} as tok)::xs
when s1 = s2 ->
change_tok tok (TIdent_Constructor (s2, i2));
aux (tok::xs)
(* recurse *)
| _::xs -> aux xs
in
aux xs
(* assumes have:
* - the typedefs
* - the right context
*
* TODO: filter the TCppDirectiveOther, have a filter_for_constructed?
*)
let find_constructed_object_and_more xs =
let rec aux xs =
match xs with
| [] -> ()
| {t=(Tdelete _| Tnew _);_}
::({t=TOCro i1} as tok1)::({t=TCCro i2} as tok2)::xs ->
change_tok tok1 (TOCro_new i1);
change_tok tok2 (TCCro_new i2);
aux xs
(* xx yy(1 ... *)
| {t=TIdent_Typedef _;_}::{t=TIdent _;_}::
({t=TOPar (ii);where=InArgument::_;_} as tok1)::xs ->
change_tok tok1 (TOPar_CplusplusInit ii);
aux xs
(* int yy(1 ... *)
| {t=tok;_}::{t=TIdent _;_}::
({t=TOPar (ii);where=InArgument::_;_} as tok1)::xs
when TH.is_basic_type tok
->
change_tok tok1 (TOPar_CplusplusInit ii);
aux xs
(* xx& yy(1 ... *)
| {t=TIdent_Typedef _;_}::{t=TAnd _}::{t=TIdent _;_}::
({t=TOPar (ii);where=InArgument::_;_} as tok1)::xs ->
change_tok tok1 (TOPar_CplusplusInit ii);
aux xs
(* xx yy(zz)
* The InArgument heuristic can't guess anything when just have
* idents inside the parenthesis. It's probably a constructed
* object though.
* TODO? could be a function declaration, especially when at Toplevel.
* If inside a function, then very probably a constructed object.
*)
| {t=TIdent_Typedef _;_}::{t=TIdent _;_}::
({t=TOPar (ii);} as tok1)::{t=TIdent _;_}::{t=TCPar _}::xs ->
change_tok tok1 (TOPar_CplusplusInit ii);
aux xs
(* xx yy(zz, ww) *)
| {t=TIdent_Typedef _;_}::{t=TIdent _;_}
::({t=TOPar (ii);} as tok1)
::{t=TIdent _;_}::{t=TComma _}::{t=TIdent _;_}
::{t=TCPar _}::xs ->
change_tok tok1 (TOPar_CplusplusInit ii);
aux xs
(* xx yy(&zz) *)
| {t=TIdent_Typedef _;_}::{t=TIdent _;_}
::({t=TOPar (ii);} as tok1)
::{t=TAnd _}
::{t=TIdent _;_}
::{t=TCPar _}::xs ->
change_tok tok1 (TOPar_CplusplusInit ii);
aux xs
(* int(), probably part of operator declaration
* could check that token before is a 'operator'
*)
| ({t=kind})::{t=TOPar _}::{t=TCPar _}::xs
when TH.is_basic_type kind ->
aux xs
(* int(...) unless it's int( * xxx ) *)
| ({t=_kind})::{t=TOPar _}::{t=TMul _}::xs ->
aux xs
| ({t=kind} as tok1)::{t=TOPar _}::xs
when TH.is_basic_type kind ->
let newone =
match kind with
| Tchar ii -> Tchar_Constr ii
| Tshort ii -> Tshort_Constr ii
| Tint ii -> Tint_Constr ii
| Tdouble ii -> Tdouble_Constr ii
| Tfloat ii -> Tfloat_Constr ii
| Tlong ii -> Tlong_Constr ii
| Tbool ii -> Tbool_Constr ii
| Tunsigned ii -> Tunsigned_Constr ii
| Tsigned ii -> Tsigned_Constr ii
| _ -> raise Impossible
in
change_tok tok1 newone;
aux xs
(* recurse *)
| _::xs -> aux xs
in
aux xs

View file

@ -0,0 +1,17 @@
val find_template_inf_sup:
Token_views_cpp.token_extended list -> unit
val find_template_commentize:
Token_views_cpp.multi_grouped list -> unit
val find_qualifier_commentize:
Token_views_cpp.token_extended list -> unit
val find_constructor_outside_class:
Token_views_cpp.token_extended list -> unit
val find_constructor:
Token_views_cpp.token_extended list -> unit
val find_constructed_object_and_more:
Token_views_cpp.token_extended list -> unit
val reclassify_tokens_before_idents_or_typedefs:
Token_views_cpp.multi_grouped list -> unit

View file

@ -0,0 +1,172 @@
(* Yoann Padioleau
*
* Copyright (C) 2002-2008 Yoann Padioleau
* Copyright (C) 2011 Facebook
*
* This program is free software; you can redistribute it and/or
* modify it under the terms of the GNU General Public License (GPL)
* version 2 as published by the Free Software Foundation.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* file license.txt for more details.
*)
open Common
open Parser_cpp
module Ast = Ast_cpp
module Parser = Parser_cpp
module TH = Token_helpers_cpp
module Hack = Parsing_hacks_lib
module PI = Parse_info
(*****************************************************************************)
(* Prelude *)
(*****************************************************************************)
(*
* To parse macro definitions I need to do some tricks
* as some information can be computed only at the lexing level. For instance
* the space after the name of the macro in '#define foo (x)' is meaningful
* but the grammar does not have this information. So define_ident() below
* look at such space and generate a special TOpar_Define token.
*
* In a similar way macro definitions can contain some antislash and newlines
* and the grammar need to know where the macro ends which is
* a line-level and so low token-level information. Hence the
* function define_line'()below and the TCommentNewline_DefineEndOfMacro.
*
* update: TCommentNewline_DefineEndOfMacro is handled in a special way
* at different places, a little bit like EOF, especially for error recovery,
* so this is an important token that should not be retagged!
*
* We also change the kind of TIdent to TIdent_Define to avoid bad interactions
* with other parsing_hack tricks. For instant if keep TIdent then
* the stringication heuristics can believe the TIdent is a string-macro.
* So simpler to change the kind of the TIdent in a macro too.
*
* ugly: maybe a better solution perhaps would be to erase
* TCommentNewline_DefineEndOfMacro from the Ast and list of tokens in parse_c.
*
* note: I do a +1 somewhere, it's for the unparsing to correctly sync.
*
* note: can't replace mark_end_define by simply a fakeInfo(). The reason
* is where is the \n TCommentSpace. Normally there is always a last token
* to synchronize on, either EOF or the token of the next toplevel.
* In the case of the #define we got in list of token
* [TCommentSpace "\n"; TDefEOL] but if TDefEOL is a fakeinfo then we will
* not synchronize on it and so we will not print the "\n".
* A solution would be to put the TDefEOL before the "\n".
*
* todo?: could put a ExpandedTok for that ?
*)
(*****************************************************************************)
(* Wrappers *)
(*****************************************************************************)
let pr2, _pr2_once = Common2.mk_pr2_wrappers Flag_parsing_cpp.verbose_lexing
(*****************************************************************************)
(* Helpers *)
(*****************************************************************************)
let mark_end_define ii =
let ii' =
{ Parse_info.
token = Parse_info.OriginTok {
(Parse_info.token_location_of_info ii) with
Parse_info.str = "";
Parse_info.charpos = PI.pos_of_info ii + 1
};
transfo = Parse_info.NoTransfo;
}
in
(* fresh_tok *) TCommentNewline_DefineEndOfMacro (ii')
let pos ii = Parse_info.string_of_info ii
(*****************************************************************************)
(* Parsing hacks for #define *)
(*****************************************************************************)
(* simple automata:
* state1 --'#define'--> state2 --change_of_line--> state1
*)
(* put the TCommentNewline_DefineEndOfMacro at the good place
* and replace \ with TCommentSpace
*)
let rec define_line_1 xs =
match xs with
| [] -> []
| (TDefine ii as x)::xs ->
let line = PI.line_of_info ii in
x::define_line_2 line ii xs
| TCppEscapedNewline ii::xs ->
pr2 (spf "WEIRD: a \\ outside a #define at %s" (pos ii));
(* fresh_tok*) TCommentSpace ii::define_line_1 xs
| x::xs ->
x::define_line_1 xs
and define_line_2 line lastinfo xs =
match xs with
| [] ->
(* should not happened, should meet EOF before *)
pr2 "PB: WEIRD in Parsing_hack_define.define_line_2";
mark_end_define lastinfo::[]
| x::xs ->
let line' = TH.line_of_tok x in
let info = TH.info_of_tok x in
(match x with
| EOF ii ->
mark_end_define lastinfo::EOF ii::define_line_1 xs
| TCppEscapedNewline ii ->
if (line' <> line)
then pr2 "PB: WEIRD: not same line number";
(* fresh_tok*) TCommentSpace ii::define_line_2 (line+1) info xs
| x ->
if line' = line
then x::define_line_2 line info xs
else
mark_end_define lastinfo::define_line_1 (x::xs)
)
(* put the TIdent_Define and TOPar_Define *)
let rec define_ident xs =
match xs with
| [] -> []
| (TDefine ii as x)::xs ->
x::
(match xs with
| (TCommentSpace _ as x)::TIdent (s,i2)::(* no space *)TOPar (i3)::xs ->
(* if TOPar_Define is just next to the ident (no space), then
* it's a macro-function. We change the token to avoid
* ambiguity between '#define foo(x)' and '#define foo (x)'
*)
x
::Hack.fresh_tok (TIdent_Define (s,i2))
::Hack.fresh_tok (TOPar_Define i3)
::define_ident xs
| (TCommentSpace _ as x)::TIdent (s,i2)::xs ->
x
::Hack.fresh_tok (TIdent_Define (s,i2))
::define_ident xs
| _ ->
pr2 (spf "WEIRD #define body, at %s" (pos ii));
define_ident xs
)
| x::xs ->
x::define_ident xs
(*****************************************************************************)
(* Entry point *)
(*****************************************************************************)
let fix_tokens_define2 xs =
define_ident (define_line_1 xs)
let fix_tokens_define a =
Common.profile_code "Hack.fix_define" (fun () -> fix_tokens_define2 a)

View file

@ -0,0 +1,6 @@
(* transform TDefine, filter the TCppEscapedNewline, generate TIdentDefine
* and other related fresh tokens.
*)
val fix_tokens_define :
Parser_cpp.token list -> Parser_cpp.token list

View file

@ -0,0 +1,287 @@
(* Yoann Padioleau
*
* Copyright (C) 2002-2008 Yoann Padioleau
* Copyright (C) 2011 Facebook
*
* This program is free software; you can redistribute it and/or
* modify it under the terms of the GNU General Public License (GPL)
* version 2 as published by the Free Software Foundation.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* file license.txt for more details.
*)
open Common
module Flag = Flag_parsing_cpp
module Ast = Ast_cpp
module TH = Token_helpers_cpp
module Parser = Parser_cpp
module PI = Parse_info
open Parser_cpp
open Token_views_cpp
(*****************************************************************************)
(* Wrappers *)
(*****************************************************************************)
let pr2, _pr2_once = Common2.mk_pr2_wrappers Flag_parsing_cpp.verbose_parsing
(*****************************************************************************)
(* Helpers *)
(*****************************************************************************)
(*
* In the following, there are some harcoded names of types or macros
* but they are not used by our heuristics! They are just here to
* enable to detect false positive by printing only the typedef/macros
* that we don't know yet. If we print everything, then we can easily
* get lost with too much verbose tracing information. So those
* functions "filter" some messages. So our heuristics are still good,
* there is no more (or not that much) hardcoded linux stuff.
*)
let msg_gen is_known printer s =
if not (!Flag.filter_msg)
then printer s
else
if not (is_known s)
then printer s
let pos ii = Parse_info.string_of_info ii
(*****************************************************************************)
(* Some debugging functions *)
(*****************************************************************************)
let pr2_pp s =
if !Flag.debug_pp
then Common.pr2 ("PP-" ^ s)
let pr2_cplusplus s =
if !Flag.debug_cplusplus
then Common.pr2 ("C++-" ^ s)
let pr2_typedef s =
if !Flag.debug_typedef
then Common.pr2 ("TYPEDEF-" ^ s)
let msg_change_tok tok =
match tok with
(* mostly in parsing_hacks_define.ml *)
| TIdent_Define (_s, _ii) ->
()
| TOPar_Define (_ii) ->
()
| TCommentNewline_DefineEndOfMacro _ ->
()
(* mostly in parsing_hacks.ml *)
| TIdent_Typedef (s, ii) ->
(* todo? also do LP.add_typedef_root s ??? *)
s +> msg_gen (fun s ->
match s with
| "u_char" | "u_short" | "u_int" | "u_long"
| "u8" | "u16" | "u32" | "u64"
| "s8" | "s16" | "s32" | "s64"
| "__u8" | "__u16" | "__u32" | "__u64"
-> true
| "acpi_handle" | "acpi_status" -> true
| "FILE" | "DIR" -> true
| s when s =~ ".*_t$" -> true
| _ -> false
)
(fun s -> pr2_typedef (spf "promoting %s at %s " s (pos ii)))
(* mostly in parsing_hacks_pp.ml *)
(* cppext: *)
| TComment_Pp (directive, ii) ->
let s = PI.str_of_info ii in
(match directive, s with
| Token_cpp.CppMacro, _ ->
pr2_pp (spf "MACRO: commented at %s" (pos ii))
| Token_cpp.CppDirective, _ when s =~ "#define.*" ->
pr2_pp (spf "DEFINE: commented at %s" (pos ii));
| Token_cpp.CppDirective, _ when s =~ "#include.*" ->
pr2_pp (spf "INCLUDE: commented at %s" (pos ii));
| Token_cpp.CppDirective, _ when s =~ "#if.*" ->
pr2_pp (spf "IFDEF: commented at %s" (pos ii));
| Token_cpp.CppDirective, _ when s =~ "#undef.*" ->
pr2_pp (spf "UNDEF: commented at %s" (pos ii));
| Token_cpp.CppDirective, _ ->
pr2_pp (spf "OTHER: commented directive at %s" (pos ii));
| _ ->
(* todo? *)
()
)
| TOBrace_DefineInit ii ->
pr2_pp (spf "DEFINE: initializer at %s" (pos ii))
| TIdent_MacroString ii ->
let s = PI.str_of_info ii in
s +> msg_gen (fun s ->
match s with
| "REVISION" | "UTS_RELEASE" | "SIZE_STR" | "DMA_STR"
-> true
(* s when s =~ ".*STR.*" -> true *)
| _ -> false
)
(fun s -> pr2_pp (spf "MACRO: string-macro %s at %s " s (pos ii)))
| TIdent_MacroStmt ii ->
pr2_pp (spf "MACRO: stmt-macro at %s" (pos ii));
| TIdent_MacroDecl (s, ii) ->
s +> msg_gen (fun s ->
match s with
| "DECLARE_MUTEX" | "DECLARE_COMPLETION" | "DECLARE_RWSEM"
| "DECLARE_WAITQUEUE" | "DECLARE_WAIT_QUEUE_HEAD"
| "DEFINE_SPINLOCK" | "DEFINE_TIMER"
| "DEVICE_ATTR" | "CLASS_DEVICE_ATTR" | "DRIVER_ATTR"
| "SENSOR_DEVICE_ATTR"
| "LIST_HEAD"
| "DECLARE_WORK" | "DECLARE_TASKLET"
| "PORT_ATTR_RO" | "PORT_PMA_ATTR"
| "DECLARE_BITMAP"
-> true
(*
| s when s =~ "^DECLARE_.*" -> true
| s when s =~ ".*_ATTR$" -> true
| s when s =~ "^DEFINE_.*" -> true
| s when s =~ "NS_DECL.*" -> true
*)
| _ -> false
)
(fun _s -> pr2_pp (spf "MACRO: macro-declare at %s" (pos ii)))
| Tconst_MacroDeclConst ii ->
pr2_pp (spf "MACRO: retag const at %s" (pos ii))
| TAny_Action ii ->
pr2_pp (spf "ACTION: retag at %s" (pos ii))
| TCPar_EOL ii ->
pr2_pp (spf "MISC: retagging ) %s" (pos ii))
(* mostly in parsing_hacks_cpp.ml *)
(* c++ext: *)
| TComment_Cpp (directive, ii) ->
let s = PI.str_of_info ii in
(match directive, s with
| Token_cpp.CplusplusTemplate, _ ->
pr2_cplusplus (spf "COM-TEMPLATE: commented at %s" (pos ii))
| Token_cpp.CplusplusQualifier, _ ->
pr2_cplusplus (spf "COM-QUALIFIER: commented at %s" (pos ii))
)
| TOPar_CplusplusInit ii ->
pr2_cplusplus (spf "constructor initializer at %s" (pos ii))
| TOCro_new ii | TCCro_new ii ->
pr2_cplusplus (spf "new [] at %s" (pos ii))
| TInf_Template ii | TSup_Template ii ->
pr2_cplusplus (spf "template <> at %s" (pos ii))
| Tchar_Constr ii | Tint_Constr ii | Tfloat_Constr ii | Tdouble_Constr ii
| Tshort_Constr ii | Tlong_Constr ii | Tbool_Constr ii
| Tunsigned_Constr ii | Tsigned_Constr ii
->
pr2_cplusplus(spf "constructed object builtin at %s" (pos ii));
| TIdent_TypedefConstr (s, ii) ->
pr2_cplusplus (spf "constructed object %s at %s" s (pos ii))
| TIdent_ClassnameInQualifier (s, ii) ->
pr2_cplusplus (spf "CLASSNAME: in qualifier context %s at %s " s (pos ii))
| TIdent_Constructor (s, ii) ->
pr2_cplusplus (spf "CONSTRUCTOR: found %s at %s " s (pos ii))
| TIdent_Templatename (s, ii) ->
pr2_cplusplus (spf "TEMPLATENAME: found %s at %s" s (pos ii))
| TColCol_BeforeTypedef ii ->
pr2_typedef (spf "RECLASSIF colcol to colcol2 at %s" (pos ii))
| TIdent_ClassnameInQualifier_BeforeTypedef (s, ii) ->
pr2_typedef (spf "RECLASSIF class in qualifier %s at %s" s (pos ii))
| TIdent_TemplatenameInQualifier_BeforeTypedef (s, ii) ->
pr2_typedef (spf "RECLASSIF template in qualifier %s at %s" s (pos ii))
| _ ->
raise Todo
let msg_context t ctx =
let ctx_str =
match ctx with
| InParameter -> "InParameter"
| InArgument -> "InArgument"
| _ -> raise Impossible
in
pr2_cplusplus (spf "CONTEXT: %s at %s" ctx_str (pos (TH.info_of_tok t)))
let change_tok extended_tok tok =
msg_change_tok tok;
(* otherwise parse_c will be lost if don't find a EOF token
* why? because paren detection had a pb because of
* some ifdef-exp?
*)
if TH.is_eof extended_tok.t
then pr2 "PB: wierd, I try to tag an EOF token as something else"
else extended_tok.t <- tok
let fresh_tok tok =
msg_change_tok tok;
tok
(* normally the caller have first filtered the set of tokens to have
* a clearer "view" to work on
*)
let set_as_comment cppkind x =
assert(not (TH.is_real_comment x.t));
change_tok x (TComment_Pp (cppkind, TH.info_of_tok x.t))
(*****************************************************************************)
(* The regexp and basic view definitions *)
(*****************************************************************************)
(*
val regexp_macro: Str.regexp
val regexp_annot: Str.regexp
val regexp_declare: Str.regexp
val regexp_foreach: Str.regexp
val regexp_typedef: Str.regexp
*)
(* opti: better to built then once and for all, especially regexp_foreach *)
let regexp_macro = Str.regexp
"^[A-Z_][A-Z_0-9]*$"
(* linuxext: *)
let regexp_declare = Str.regexp
".*DECLARE.*"
(* firefoxext: *)
let regexp_ns_decl_like = Str.regexp
("\\(" ^
"NS_DECL_\\|NS_DECLARE_\\|NS_IMPL_\\|" ^
"NS_IMPLEMENT_\\|NS_INTERFACE_\\|NS_FORWARD_\\|NS_HTML_\\|" ^
"NS_DISPLAY_\\|NS_IMPL_\\|" ^
"TX_DECL_\\|DOM_CLASSINFO_\\|NS_CLASSINFO_\\|IMPL_INTERNAL_\\|" ^
"ON_\\|EVT_\\|NS_UCONV_\\|NS_GENERIC_\\|NS_COM_" ^
"\\).*")

View file

@ -0,0 +1,20 @@
val pr2_pp: string -> unit
val set_as_comment:
Token_cpp.cppcommentkind -> Token_views_cpp.token_extended -> unit
val msg_context:
Parser_cpp.token -> Token_views_cpp.context -> unit
val change_tok:
Token_views_cpp.token_extended -> Parser_cpp.token -> unit
val fresh_tok:
Parser_cpp.token -> Parser_cpp.token
val regexp_ns_decl_like: Str.regexp
val regexp_macro: Str.regexp
val regexp_declare: Str.regexp

View file

@ -0,0 +1,755 @@
(* Yoann Padioleau
*
* Copyright (C) 2002-2008 Yoann Padioleau
* Copyright (C) 2011 Facebook
*
* This program is free software; you can redistribute it and/or
* modify it under the terms of the GNU General Public License (GPL)
* version 2 as published by the Free Software Foundation.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* file license.txt for more details.
*)
open Common
module Flag = Flag_parsing_cpp
module Ast = Ast_cpp
module TH = Token_helpers_cpp
module TV = Token_views_cpp
module Parser = Parser_cpp
module PI = Parse_info
open Parser_cpp
open Token_views_cpp
open Parsing_hacks_lib
(*****************************************************************************)
(* Prelude *)
(*****************************************************************************)
(*
* This file gathers parsing heuristics related to the C preprocessor cpp.
*)
(*****************************************************************************)
(* Helpers *)
(*****************************************************************************)
let (==~) = Common2.(==~)
(* the pair is the status of '()' and '{}', ex: (-1,0)
* if too much ')' and good '{}'
* could do for [] too ?
* could do for ',' if encounter ',' at "toplevel", not inside () or {}
* then if have ifdef, then certainly can lead to a problem.
*)
let (count_open_close_stuff_ifdef_clause: ifdef_grouped list -> (int * int)) =
fun xs ->
let cnt_paren, cnt_brace = ref 0, ref 0 in
xs +> iter_token_ifdef (fun x ->
(match x.t with
| x when TH.is_opar x -> incr cnt_paren
| x when TH.is_obrace x -> incr cnt_brace
| x when TH.is_cpar x -> decr cnt_paren
| x when TH.is_obrace x -> decr cnt_brace
| _ -> ()
)
);
!cnt_paren, !cnt_brace
(* look if there is a '{' just after the closing ')', and handling the
* possibility to have nested expressions inside nested parenthesis
*)
(*
let is_really_foreach xs =
let rec is_foreach_aux = function
| [] -> false, []
| TCPar _::TOBrace _::xs -> true, xs
(* the following attempts to handle the cases where there is a
single statement in the body of the loop. undoubtedly more
cases are needed.
todo: premier(statement) - suivant(funcall)
*)
| TCPar _::TIdent _::xs -> true, xs
| TCPar _::Tif _::xs -> true, xs
| TCPar _::Twhile _::xs -> true, xs
| TCPar _::Tfor _::xs -> true, xs
| TCPar _::Tswitch _::xs -> true, xs
| TCPar _::xs -> false, xs
| TOPar _::xs ->
let (_, xs') = is_foreach_aux xs in
is_foreach_aux xs'
| x::xs -> is_foreach_aux xs
in
is_foreach_aux xs +> fst
*)
(* TODO: set_ifdef_parenthize_info ?? from parsing_c/ *)
let filter_pp_or_comment_stuff xs =
let rec aux xs =
match xs with
| [] -> []
| x::xs ->
(match x.TV.t with
| tok when TH.is_comment tok ->
aux xs
(* don't want drop the define, or if drop, have to drop
* also its body otherwise the line heuristics may be lost
* by not finding the TDefine in column 0 but by finding
* a TDefineIdent in a column > 0
*
* todo? but define often contain some unbalanced {
*)
| Parser.TDefine _ ->
x::aux xs
| tok when TH.is_pp_instruction tok ->
aux xs
| _ ->
x::aux xs
)
in
aux xs
(*****************************************************************************)
(* Ifdef keeping/passing *)
(*****************************************************************************)
(* #if 0, #if 1, #if LINUX_VERSION handling *)
let rec find_ifdef_bool xs =
xs +> List.iter (function
| NotIfdefLine _ -> ()
| Ifdefbool (is_ifdef_positif, xxs, info_ifdef_stmt) ->
if is_ifdef_positif
then pr2_pp "commenting parts of a #if 1 or #if LINUX_VERSION"
else pr2_pp "commenting a #if 0 or #if LINUX_VERSION or __cplusplus";
(match xxs with
| [] -> raise Impossible
| firstclause::xxs ->
info_ifdef_stmt +> List.iter (set_as_comment Token_cpp.CppDirective);
if is_ifdef_positif
then xxs +> List.iter
(iter_token_ifdef (set_as_comment Token_cpp.CppOther))
else begin
firstclause +> iter_token_ifdef (set_as_comment Token_cpp.CppOther);
(match List.rev xxs with
(* keep only last *)
| _last::startxs ->
startxs +> List.iter
(iter_token_ifdef (set_as_comment Token_cpp.CppOther))
| [] -> (* not #else *) ()
);
end
);
| Ifdef (xxs, _info_ifdef_stmt) -> xxs +> List.iter find_ifdef_bool
)
let thresholdIfdefSizeMid = 6
(* infer ifdef involving not-closed expressions/statements *)
let rec find_ifdef_mid xs =
xs +> List.iter (function
| NotIfdefLine _ -> ()
| Ifdef (xxs, info_ifdef_stmt) ->
(match xxs with
| [] -> raise Impossible
| [_first] -> ()
| _first::second::rest ->
(* don't analyse big ifdef *)
if xxs +> List.for_all
(fun xs -> List.length xs <= thresholdIfdefSizeMid) &&
(* don't want nested ifdef *)
xxs +> List.for_all (fun xs ->
xs +> List.for_all
(function NotIfdefLine _ -> true | _ -> false)
)
then
let counts = xxs +> List.map count_open_close_stuff_ifdef_clause in
let cnt1, cnt2 = List.hd counts in
if cnt1 <> 0 || cnt2 <> 0
(*???? && counts +> List.for_all (fun x -> x = (cnt1, cnt2)) *)
(*
if counts +> List.exists (fun (cnt1, cnt2) ->
cnt1 <> 0 || cnt2 <> 0
)
*)
then begin
pr2_pp "found ifdef-mid-something";
(* keep only first, treat the rest as comment *)
info_ifdef_stmt +> List.iter (set_as_comment Token_cpp.CppDirective);
(second::rest) +> List.iter
(iter_token_ifdef (set_as_comment Token_cpp.CppOther));
end
);
List.iter find_ifdef_mid xxs
(* no need complex analysis for ifdefbool *)
| Ifdefbool (_, xxs, _info_ifdef_stmt) ->
List.iter find_ifdef_mid xxs
)
let thresholdFunheaderLimit = 4
(* ifdef defining alternate function header, type *)
let rec find_ifdef_funheaders = function
| [] -> ()
| NotIfdefLine _::xs -> find_ifdef_funheaders xs
(* ifdef-funheader if ifdef with 2 lines and a '{' in next line *)
| Ifdef
([(NotIfdefLine (({col = 0} as _xline1)::_line1))::ifdefblock1;
(NotIfdefLine (({col = 0} as xline2)::line2))::ifdefblock2
], info_ifdef_stmt
)
::NotIfdefLine (({t=TOBrace _i; col = 0})::_line3)
::xs
when List.length ifdefblock1 <= thresholdFunheaderLimit &&
List.length ifdefblock2 <= thresholdFunheaderLimit
->
find_ifdef_funheaders xs;
info_ifdef_stmt +> List.iter (set_as_comment Token_cpp.CppDirective);
let all_toks = [xline2] @ line2 in
all_toks +> List.iter (set_as_comment Token_cpp.CppOther) ;
ifdefblock2 +> iter_token_ifdef (set_as_comment Token_cpp.CppOther);
(* ifdef with nested ifdef *)
| Ifdef
([[NotIfdefLine (({col = 0} as _xline1)::_line1)];
[Ifdef
([[NotIfdefLine (({col = 0} as xline2)::line2)];
[NotIfdefLine (({col = 0} as xline3)::line3)];
], info_ifdef_stmt2
)
]
], info_ifdef_stmt
)
::NotIfdefLine (({t=TOBrace _i; col = 0})::_line4)
::xs
->
find_ifdef_funheaders xs;
info_ifdef_stmt +> List.iter (set_as_comment Token_cpp.CppDirective);
info_ifdef_stmt2 +> List.iter (set_as_comment Token_cpp.CppDirective);
let all_toks = [xline2;xline3] @ line2 @ line3 in
all_toks +> List.iter (set_as_comment Token_cpp.CppOther);
(* ifdef with elseif *)
| Ifdef
([[NotIfdefLine (({col = 0} as _xline1)::_line1)];
[NotIfdefLine (({col = 0} as xline2)::line2)];
[NotIfdefLine (({col = 0} as xline3)::line3)];
], info_ifdef_stmt
)
::NotIfdefLine (({t=TOBrace _i; col = 0})::_line4)
::xs
->
find_ifdef_funheaders xs;
info_ifdef_stmt +> List.iter (set_as_comment Token_cpp.CppDirective);
let all_toks = [xline2;xline3] @ line2 @ line3 in
all_toks +> List.iter (set_as_comment Token_cpp.CppOther)
| Ifdef (xxs,_)::xs
| Ifdefbool (_, xxs,_)::xs ->
List.iter find_ifdef_funheaders xxs;
find_ifdef_funheaders xs
(*
let adjust_inifdef_include xs =
xs +> List.iter (function
| NotIfdefLine _ -> ()
| Ifdef (xxs, info_ifdef_stmt) | Ifdefbool (_, xxs, info_ifdef_stmt) ->
xxs +> List.iter (iter_token_ifdef (fun tokext ->
match tokext.t with
| Parser.TInclude (s1, s2, ii) ->
(* todo: inifdef_ref := true; *)
()
| _ -> ()
));
)
*)
(*****************************************************************************)
(* Builtin macros using standard.h or other defs *)
(*****************************************************************************)
(* now in pp_token.ml *)
(*****************************************************************************)
(* Stringification *)
(*****************************************************************************)
let rec find_string_macro_paren xs =
match xs with
| [] -> ()
| Parenthised(xxs, _)::xs ->
xxs +> List.iter (fun xs ->
if xs +> List.exists
(function PToken({t=TString _}) -> true | _ -> false) &&
xs +> List.for_all
(function PToken({t=TString _}) | PToken({t=TIdent _}) ->
true | _ -> false)
then
xs +> List.iter (fun tok ->
match tok with
| PToken({t=TIdent (_s,_)} as id) ->
change_tok id (TIdent_MacroString (TH.info_of_tok id.t))
| _ -> ()
)
else
find_string_macro_paren xs
);
find_string_macro_paren xs
| PToken _ ::xs ->
find_string_macro_paren xs
(*****************************************************************************)
(* Macros *)
(*****************************************************************************)
(* don't forget to recurse in each case.
* note that the code below is called after the ifdef phase simplification,
* so if this previous phase is buggy, then it may pass some code that
* could be matched by the following rules but will not.
**)
let rec find_macro_paren xs =
match xs with
| [] -> ()
(* attribute *)
| PToken ({t=Tattribute _} as id)
::Parenthised (xxs,info_parens)
::xs
->
pr2_pp ("MACRO: __attribute detected ");
[Parenthised (xxs, info_parens)] +>
iter_token_paren (set_as_comment Token_cpp.CppAttr);
set_as_comment Token_cpp.CppAttr id;
find_macro_paren xs
(* stringification
*
* the order of the matching clause is important
*
*)
(* string macro with params, before case *)
| PToken ({t=TString _})::PToken ({t=TIdent (_s,_)} as id)
::Parenthised (xxs, info_parens)
::xs ->
change_tok id (TIdent_MacroString (TH.info_of_tok id.t));
[Parenthised (xxs, info_parens)] +>
iter_token_paren (set_as_comment Token_cpp.CppMacro);
find_macro_paren xs
(* after case *)
| PToken ({t=TIdent (_s,_)} as id)
::Parenthised (xxs, info_parens)
::PToken ({t=TString _})
::xs ->
change_tok id (TIdent_MacroString (TH.info_of_tok id.t));
[Parenthised (xxs, info_parens)] +>
iter_token_paren (set_as_comment Token_cpp.CppMacro);
find_macro_paren xs
(* for the case where the string is not inside a funcall, but
* for instance in an initializer.
*)
(* string macro variable, before case *)
| PToken ({t=TString ((str,_),_)})::PToken ({t=TIdent (_s,_)} as id)
::xs ->
(* c++ext: *)
if str <> "C" then begin
change_tok id (TIdent_MacroString (TH.info_of_tok id.t));
find_macro_paren xs
end
(* bugfix, forgot to recurse in else case too ... *)
else
find_macro_paren xs
(* after case *)
| PToken ({t=TIdent (_s,_)} as id)::PToken ({t=TString _})
::xs ->
change_tok id (TIdent_MacroString (TH.info_of_tok id.t));
find_macro_paren xs
(* TODO: cooperating with standard.h *)
| PToken ({t=TIdent (s,_i1)} as id)::xs
when s = "MACROSTATEMENT" ->
change_tok id (TIdent_MacroStmt(TH.info_of_tok id.t));
find_macro_paren xs
(* recurse *)
| (PToken _x)::xs -> find_macro_paren xs
| (Parenthised (xxs, _))::xs ->
xxs +> List.iter find_macro_paren;
find_macro_paren xs
(* don't forget to recurse in each case *)
let rec find_macro_lineparen xs =
match xs with
| [] -> ()
(* firefoxext: ex: NS_DECL_NSIDOMNODELIST *)
| (Line ([PToken ({t=TIdent (s,_)} as macro);]))::xs
when s ==~ regexp_ns_decl_like ->
set_as_comment Token_cpp.CppMacro macro;
find_macro_lineparen (xs)
(* firefoxext: ex: NS_DECL_NSIDOMNODELIST; *)
| (Line ([PToken ({t=TIdent (s,_)} as macro);
PToken ({t=TPtVirg _})]))::xs
when s ==~ regexp_ns_decl_like ->
set_as_comment Token_cpp.CppMacro macro;
find_macro_lineparen (xs)
(* firefoxext: ex: NS_IMPL_XXX(a) *)
| (Line ([PToken ({t=TIdent (s,_)} as macro);
Parenthised (xxs,info_parens);
]))::xs
when s ==~ regexp_ns_decl_like ->
[Parenthised (xxs, info_parens)] +>
iter_token_paren (set_as_comment Token_cpp.CppMacro);
set_as_comment Token_cpp.CppMacro macro;
find_macro_lineparen (xs)
(* linuxext: ex: static [const] DEVICE_ATTR(); *)
| (Line
(
[PToken ({t=Tstatic _});
PToken ({t=TIdent (s,_)} as macro);
Parenthised (_xxs,_);
PToken ({t=TPtVirg _});
]
))::xs
when (s ==~ regexp_macro) ->
let info = TH.info_of_tok macro.t in
change_tok macro (TIdent_MacroDecl (PI.str_of_info info, info));
find_macro_lineparen (xs)
(* the static const case *)
| (Line
(
[PToken ({t=Tstatic _});
PToken ({t=Tconst _} as const);
PToken ({t=TIdent (s,_)} as macro);
Parenthised (_xxs,_info_parens);
PToken ({t=TPtVirg _});
]
(*as line1*)
))
::xs
when (s ==~ regexp_macro) ->
let info = TH.info_of_tok macro.t in
change_tok macro (TIdent_MacroDecl (PI.str_of_info info, info));
(* need retag this const, otherwise ambiguity in grammar
21: shift/reduce conflict (shift 121, reduce 137) on Tconst
decl2 : Tstatic . TMacroDecl TOPar argument_list TCPar ...
decl2 : Tstatic . Tconst TMacroDecl TOPar argument_list TCPar ...
storage_class_spec : Tstatic . (137)
*)
change_tok const (Tconst_MacroDeclConst (TH.info_of_tok const.t));
find_macro_lineparen (xs)
(* same but without trailing ';'
*
* I do not put the final ';' because it can be on a multiline and
* because of the way mk_line is coded, we will not have access to
* this ';' on the next line, even if next to the ')' *)
| (Line
([PToken ({t=Tstatic _});
PToken ({t=TIdent (s,_)} as macro);
Parenthised (_xxs,_);
]
))::xs
when s ==~ regexp_macro ->
let info = TH.info_of_tok macro.t in
change_tok macro (TIdent_MacroDecl (PI.str_of_info info, info));
find_macro_lineparen (xs)
(* on multiple lines *)
| (Line
(
(PToken ({t=Tstatic _})::[]
)))
::(Line
(
[PToken ({t=TIdent (s,_)} as macro);
Parenthised (_,_);
PToken ({t=TPtVirg _});
]
)
)::xs
when (s ==~ regexp_macro) ->
let info = TH.info_of_tok macro.t in
change_tok macro (TIdent_MacroDecl (PI.str_of_info info, info));
find_macro_lineparen xs
(* linuxext: ex: DECLARE_BITMAP();
*
* Here I use regexp_declare and not regexp_macro because
* Sometimes it can be a FunCallMacro such as DEBUG(foo());
* Here we don't have the preceding 'static' so only way to
* not have positive is to restrict to .*DECLARE.* macros.
*
* but there is a grammar rule for that, so don't need this case anymore
* unless the parameter of the DECLARE_xxx are wierd and can not be mapped
* on a argument_list
*)
| (Line
([PToken ({t=TIdent (s,_)} as macro);
Parenthised (_,_);
PToken ({t=TPtVirg _});
]
))::xs
when (s ==~ regexp_declare) ->
let info = TH.info_of_tok macro.t in
change_tok macro (TIdent_MacroDecl (PI.str_of_info info, info));
find_macro_lineparen xs
(* toplevel macros.
* module_init(xxx)
*
* Could also transform the TIdent in a TMacroTop but can have false
* positive, so easier to just change the TCPar and so just solve
* the end-of-stream pb of ocamlyacc
*)
| (Line
([PToken ({t=TIdent (_s,_ii); col = col1; where = ctx} as _macro);
Parenthised (_,info_parens);
] as _line1
))
::xs when col1 = 0
->
let condition =
(* to reduce number of false positive *)
(match xs with
| (Line (PToken ({col = col2 } as other)::_restline2))::_ ->
TH.is_eof other.t || (col2 = 0 &&
(match other.t with
| TOBrace _ -> false (* otherwise would match funcdecl *)
| TCBrace _ when List.hd ctx <> InFunction -> false
| TPtVirg _
| TCol _
-> false
| tok when TH.is_binary_operator tok -> false
| _ -> true
)
)
| _ -> false
)
in
if condition
then begin
(* just to avoid the end-of-stream pb of ocamlyacc *)
let tcpar = Common2.list_last info_parens in
change_tok tcpar (TCPar_EOL (TH.info_of_tok tcpar.t));
(*macro.t <- TMacroTop (s, TH.info_of_tok macro.t);*)
end;
find_macro_lineparen xs
(* macro with parameters
* ex: DEBUG()
* return x;
*)
| (Line
([PToken ({t=TIdent (_s,_ii); col = col1; where = ctx} as macro);
Parenthised (xxs,info_parens);
] as _line1
))
::(Line
(PToken ({col = col2 } as other)::_restline2
) as line2)
::xs
(* when s ==~ regexp_macro *)
->
let condition =
(col1 = col2 &&
(match other.t with
| TOBrace _ -> false (* otherwise would match funcdecl *)
| TCBrace _ when List.hd ctx <> InFunction -> false
| TPtVirg _
| TCol _
-> false
| tok when TH.is_binary_operator tok -> false
| _ -> true
)
)
||
(col2 <= col1 &&
(match other.t with
| TCBrace _ when List.hd ctx = InFunction -> true
| Treturn _ -> true
| Tif _ -> true
| Telse _ -> true
| _ -> false
)
)
in
if condition
then
if col1 = 0 then ()
else begin
change_tok macro (TIdent_MacroStmt (TH.info_of_tok macro.t));
[Parenthised (xxs, info_parens)] +>
iter_token_paren (set_as_comment Token_cpp.CppMacro);
end;
find_macro_lineparen (line2::xs)
(* linuxext:? single macro
* ex: LOCK
* foo();
* UNLOCK
*)
| (Line
([PToken ({t=TIdent (_s,_ii); col = col1; where = ctx} as macro);
] as _line1
))
::(Line
(PToken ({col = col2 } as other)::_restline2
) as line2)
::xs ->
(* when s ==~ regexp_macro *)
let condition =
(col1 = col2 &&
col1 <> 0 && (* otherwise can match typedef of fundecl*)
(match other.t with
| TPtVirg _ -> false
| TOr _ -> false
| TCBrace _ when List.hd ctx <> InFunction -> false
| tok when TH.is_binary_operator tok -> false
| _ -> true
)) ||
(col2 <= col1 &&
(match other.t with
| TCBrace _ when List.hd ctx = InFunction -> true
| Treturn _ -> true
| Tif _ -> true
| Telse _ -> true
| _ -> false
))
in
if condition
then change_tok macro (TIdent_MacroStmt (TH.info_of_tok macro.t));
find_macro_lineparen (line2::xs)
| _x::xs ->
find_macro_lineparen xs
(*****************************************************************************)
(* #Define tobrace init *)
(*****************************************************************************)
let is_init tok2 tok3 =
match tok2.t, tok3.t with
| TInt _, TComma _ -> true
| TString _, TComma _ -> true
| TIdent _, TComma _ -> true
| _ -> false
let find_define_init_brace_paren xs =
let rec aux xs =
match xs with
| [] -> ()
(* mainly for firefox *)
| (PToken {t=TDefine _})
::(PToken {t=TIdent_Define (_s,_)})
::(PToken ({t=TOBrace i1} as tokbrace))
::(PToken tok2)
::(PToken tok3)
::xs ->
if is_init tok2 tok3
then change_tok tokbrace (TOBrace_DefineInit i1);
aux xs
(* mainly for linux, especially in sound/ *)
| (PToken {t=TDefine _})
::(PToken {t=TIdent_Define (s,_); col=c})
::(Parenthised(_, {col=c2; _}::_))
::(PToken ({t=TOBrace i1} as tokbrace))
::(PToken tok2)
::(PToken tok3)
::xs when c2 = c + String.length s ->
if is_init tok2 tok3
then change_tok tokbrace (TOBrace_DefineInit i1);
aux xs
(* ugly: for plan9, too general? *)
| (PToken {t=TDefine _})
::(PToken {t=TIdent_Define (_s,_)})
::(Parenthised(_xxx, _))
::(PToken ({t=TOBrace i1} as tokbrace))
(* can be more complex expression than just an int, like (b)&... *)
::(Parenthised(_, _))
::(PToken {t=(TAnd _|TOr _);_})
::xs ->
change_tok tokbrace (TOBrace_DefineInit i1);
aux xs
(* recurse *)
| (PToken _)::xs -> aux xs
| (Parenthised (_, _))::xs ->
(* not need for tobrace init:
* xxs +> List.iter aux;
*)
aux xs
in
aux xs

View file

@ -0,0 +1,20 @@
val find_ifdef_funheaders:
Token_views_cpp.ifdef_grouped list -> unit
val find_ifdef_bool:
Token_views_cpp.ifdef_grouped list -> unit
val find_ifdef_mid:
Token_views_cpp.ifdef_grouped list -> unit
val find_define_init_brace_paren:
Token_views_cpp.paren_grouped list -> unit
val find_string_macro_paren:
Token_views_cpp.paren_grouped list -> unit
val find_macro_lineparen:
Token_views_cpp.paren_grouped Token_views_cpp.line_grouped list -> unit
val find_macro_paren:
Token_views_cpp.paren_grouped list -> unit
val filter_pp_or_comment_stuff:
Token_views_cpp.token_extended list -> Token_views_cpp.token_extended list

View file

@ -0,0 +1,375 @@
(* Yoann Padioleau
*
* Copyright (C) 2011,2014 Facebook
* Copyright (C) 2002-2008 Yoann Padioleau
*
* This program is free software; you can redistribute it and/or
* modify it under the terms of the GNU General Public License (GPL)
* version 2 as published by the Free Software Foundation.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* file license.txt for more details.
*)
open Common
module TV = Token_views_cpp
module TH = Token_helpers_cpp
module Ast = Ast_cpp
open Parser_cpp
open Token_views_cpp
open Parsing_hacks_lib
(*****************************************************************************)
(* Prelude *)
(*****************************************************************************)
(*
* This file gathers parsing heuristics related to the typedefs.
* C does not have a context-free grammar; C requires the parser to know when
* an ident corresponds to a typedef or an ident. This normally means that
* we must call cpp on the file and have the lexer and parser cooperate
* to remember what is what. In lang_cpp/ we want to parse as-is,
* which means we need to infer back whether an identifier is
* a typedef or not.
*
* In this module we use a view that is more convenient for
* typedefs detection. We got rid of:
* - template arguments (see find_template_commentize())
* - qualifiers (see find_qualifier_commentize)
* - differences between & and * (filter_for_typedef() below)
* - differences between TIdent and TOperator,
* - const, volatile, restrict keywords
* - TODO merge multiple ** or *& or whatever
*
* history:
* - We used to make the lexer and parser cooperate in a lexerParser.ml file
* - this was not enough because of declarations such as 'acpi acpi;'
* and so we had to enable/disable the ident->typedef mechanism
* which requires even more lexer/parser cooperation
* - this was ugly too so now we use a typedef "inference" mechanism
* - we refined the typedef inference to sometimes use InParameter hint
* and more contextual information from token_views_context.ml
*)
(*****************************************************************************)
(* Helpers *)
(*****************************************************************************)
let look_like_multiplication_context tok_before =
match tok_before with
| TEq _ | TAssign _
| TWhy _
| Treturn _
| TDot _ | TPtrOp _ | TPtrOpStar _ | TDotStar _
| TOCro _
-> true
| tok when TH.is_binary_operator_except_star tok -> true
| _ -> false
let look_like_declaration_context tok_before =
match tok_before with
| TOBrace _
| TPtVirg _
| TCommentNewline_DefineEndOfMacro _
| TInclude _
(* no!! | TCBrace _, I think because of nested struct so can have
* struct { ... } v;
*)
-> true
| _ when TH.is_privacy_keyword tok_before -> true
| _ -> false
let fakeInfo = { Parse_info.
token = Parse_info.FakeTokStr ("",None);
transfo = Parse_info.NoTransfo;
}
(*****************************************************************************)
(* Better View *)
(*****************************************************************************)
let filter_for_typedef multi_groups =
(* a sentinel, which helps a few typedef heuristics which look
* for a token before which would not work for the first toplevel
* declaration.
*)
let multi_groups =
Tok(mk_token_fake (TPtVirg (fakeInfo)))::multi_groups in
let _template_args = ref [] in
(* remove template and other things
* less: right now this is less useful because we actually
* comment template args in a previous pass, but at some point this
* will be useful.
*)
let rec aux xs =
xs +> Common.map_filter (function
| TV.Angle (_, _, _) ->
(* todo: analayze xs!! add in _template_args
* todo: add the t1,t2 around xs to have
* some sentinel for the typedef heuristics patterns
* who often look for the token just before the typedef.
*)
None
| TV.Braces (t1, xs, t2) ->
Some (TV.Braces (t1, aux xs, t2))
| TV.Parens (t1, xs, t2) ->
Some (TV.Parens (t1, aux xs, t2))
(* remove other noise for the typedef inference *)
| TV.Tok t1 ->
match t1.TV.t with
(* const is a strong signal for having a typedef, so why skip it?
* because it forces to duplicate rules. We need to infer
* the type anyway even when there is no const around.
* todo? maybe could do a special pass first that infer typedef
* using only const rules, and then remove those const so
* have best of both worlds.
*)
| Tconst _ | Tvolatile _
| Trestrict _
-> None
| Tregister _ | Tstatic _ | Tauto _ | Textern _
| Ttypedef _
-> None
| Tvirtual _ | Tfriend _ | Tinline _ | Tmutable _
-> None
(* let's transform all '&' into '*'
* todo: need propagate also the where?
*)
| TAnd ii -> Some (TV.Tok (mk_token_extended (TMul ii)))
(* and operator into TIdent
* TODO: skip the token just after the operator keyword?
* could help some heuristics too
*)
| Toperator ii ->
Some (TV.Tok (mk_token_extended (TIdent ("operator", ii))))
| _ -> Some (TV.Tok t1)
)
in
let xs = aux multi_groups in
(* todo: look also for _template_args *)
[TV.tokens_of_multi_grouped xs]
(*****************************************************************************)
(* Main heuristics *)
(*****************************************************************************)
(*
* Below we assume a view without:
* - comments and cpp-directives
* - template stuff and qualifiers (but not TIdent_ClassnameAsQualifier)
* - const/volatile/restrict
* - & => *
*
* With such a view we can write less patterns.
*
* Note that qualifiers are slightly less important to filter because
* most of the heuristics below look for tokens after the ident
* and qualifiers are usually before.
*
* todo: do it on multi view? all those rules with TComma and TOPar
* are ugly.
*)
let find_typedefs xxs =
let rec aux xs =
match xs with
| [] -> ()
(* struct x ...
* those identifiers (called tags) must not be transformed in typedefs *)
| {t=(Tstruct _ | Tunion _ | Tenum _ | Tclass _)}::{t=TIdent _}::xs ->
aux xs
(* xx yy *)
| ({t=TIdent (s,i1)} as tok1)::{t=TIdent _}::xs ->
change_tok tok1 (TIdent_Typedef (s, i1));
aux xs
(* xx ( *yy )( *)
| ({t=TIdent (s,i1)} as tok1)::{t=TOPar _}::{t=TMul _}
::{t=TIdent _}::{t=TCPar _}::({t=TOPar _} as tok2)::xs ->
change_tok tok1 (TIdent_Typedef (s, i1));
aux (tok2::xs)
(* xx* ( *yy )( *)
| ({t=TIdent (s,i1)} as tok1)::{t=TMul _}::{t=TOPar _}::{t=TMul _}
::{t=TIdent _}::{t=TCPar _}::({t=TOPar _} as tok2)::xs ->
change_tok tok1 (TIdent_Typedef (s, i1));
aux (tok2::xs)
(* xx ( *yy[x] )( *)
| ({t=TIdent (s,i1)} as tok1)::{t=TOPar _}::{t=TMul _}
::{t=TIdent _}::{t=TOCro _}::_::{t=TCCro _}::{t=TCPar _}::({t=TOPar _} as tok2)::xs ->
change_tok tok1 (TIdent_Typedef (s, i1));
aux (tok2::xs)
(* xx* ( *yy[x] )( *)
| ({t=TIdent (s,i1)} as tok1)::{t=TMul _}::{t=TOPar _}::{t=TMul _}
::{t=TIdent _}::{t=TOCro _}::_::{t=TCCro _}::{t=TCPar _}::({t=TOPar _} as tok2)::xs ->
change_tok tok1 (TIdent_Typedef (s, i1));
aux (tok2::xs)
(* xx ( *yy[]) *)
| ({t=TIdent (s,i1)} as tok1)::{t=TOPar _}::{t=TMul _}
::{t=TIdent _}::{t=TOCro _}::{t=TCCro _}::{t=TCPar _}::xs ->
change_tok tok1 (TIdent_Typedef (s, i1));
aux xs
(* + xx * yy *)
| {t=tok_before}::{t=TIdent (_s,_)}::{t=TMul _}::{t=TIdent _}::xs
when look_like_multiplication_context tok_before ->
aux xs
(* { xx * yy *)
| {t=tok_before}::({t=TIdent (s,i1)} as tok1)::{t=TMul _}::{t=TIdent _}::xs
when look_like_declaration_context tok_before ->
change_tok tok1 (TIdent_Typedef (s, i1));
aux xs
(* } xx * yy *)
(* because TCBrace is not anymore in look_like_declaration_context *)
| {t=TCBrace _}::({t=TIdent (s,i1)} as tok1)::{t=TMul _}::{t=TIdent _}::xs ->
change_tok tok1 (TIdent_Typedef (s, i1));
aux xs
(* xx * yy
* could be a multiplication too, so need InParameter guard/
* less: the InParameter has some FPs, so maybe better to
* rely on the spacing hint, see the rule below.
*)
| ({t=TIdent (s,i1);where=InParameter::_} as tok1)::{t=TMul _}
::{t=TIdent _}::xs
->
change_tok tok1 (TIdent_Typedef (s, i1));
aux xs
(* xx *yy *)
| ({t=TIdent (s,i1);col=c0} as tok1)::{t=TMul _;col=c1}::{t=TIdent _;col=c2}::xs
when c2 = c1 + 1 && c1 >= c0 + String.length s + 1
->
change_tok tok1 (TIdent_Typedef (s, i1));
aux xs
(* xx* yy *)
| ({t=TIdent(s,i1);col=c0}as tok1)::{t=TMul _;col=c1}::{t=TIdent _;col=c2}::xs
when c1 = c0 + String.length s && c2 >= c1 + 2
->
change_tok tok1 (TIdent_Typedef (s, i1));
aux xs
(* xx ** yy
* less could be a multiplication too, but with less probability
*)
| ({t=TIdent (s,i1)} as tok1)::{t=TMul _}::{t=TMul _}::{t=TIdent _}::xs ->
change_tok tok1 (TIdent_Typedef (s, i1));
aux xs
(* (xx) yy and not a if/while before '(' (and yy can also be a constant) *)
| {t=tok1}::{t=TOPar _}::({t=TIdent(s, i1)} as tok3)::{t=TCPar _}
::{t = (TIdent _|TInt _|TString _|TFloat _|TTilde _|TOPar _) }::xs
when not (TH.is_stuff_taking_parenthized tok1) (* && line are the same?*)->
change_tok tok3 (TIdent_Typedef (s, i1));
(* todo? recurse on bigger ? *)
aux xs
(* todo: = (xx) ..., |= (xx) ..., (xx)~, ... *)
(* (xx){ gccext: kenccext: *)
| {t=tok1}::{t=TOPar _}::({t=TIdent(s, i1)} as tok3)::{t=TCPar _}
::({t=TOBrace _} as tok5)::xs
when not (TH.is_stuff_taking_parenthized tok1) ->
change_tok tok3 (TIdent_Typedef (s, i1));
aux (tok5::xs)
(* (xx * ), not that pointer function are ( *xx ), so star before.
* TODO: does not really need the closing paren?
* TODO: check that not InParameter or InArgument?
*)
| {t=TOPar _}::({t=TIdent(s, i1)} as tok3)::{t=TMul _}::{t=TCPar _}::xs ->
change_tok tok3 (TIdent_Typedef (s, i1));
aux xs
(* (xx ** ) *)
| {t=TOPar _}::({t=TIdent(s, i1)} as tok3)
::{t=TMul _}::{t=TMul _}::{t=TCPar _}::xs ->
change_tok tok3 (TIdent_Typedef (s, i1));
aux xs
(* xx* [,)]
* don't forget to recurse by reinjecting the comma or closing paren
*)
| ({t=TIdent(s, i1)} as tok1)::{t=TMul _}
::({t=(TComma _| TCPar _)} as x)::xs ->
change_tok tok1 (TIdent_Typedef (s, i1));
aux (x::xs)
(* xx** [,)] *)
| ({t=TIdent(s, i1)} as tok1)::{t=TMul _}::{t=TMul _}
::({t=(TComma _| TCPar _)} as x)::xs ->
change_tok tok1 (TIdent_Typedef (s, i1));
aux (x::xs)
(* xx*** [,)] *)
| ({t=TIdent(s, i1)} as tok1)::{t=TMul _}::{t=TMul _}::{t=TMul _}
::({t=(TComma _| TCPar _)} as x)::xs ->
change_tok tok1 (TIdent_Typedef (s, i1));
aux (x::xs)
(* xx*[] [,)] *)
| ({t=TIdent(s, i1)} as tok1)::{t=TMul _}::{t=TOCro _}::{t=TCCro _}
::({t=(TComma _| TCPar _)} as x)::xs ->
change_tok tok1 (TIdent_Typedef (s, i1));
aux (x::xs)
(* [(,] xx [),] where InParameter *)
(* hmmm: todo: some false positives on InParameter, see mini/constants.c,
* so now simpler to add a TIdent in the parameter_decl rule
*)
| {t=(TOPar _ | TComma _)}::({t=TIdent (s, i1); where=InParameter::_} as tok1)
::({t=(TCPar _ | TComma _)} as tok2)::xs ->
change_tok tok1 (TIdent_Typedef (s, i1));
aux (tok2::xs)
(* [(,] xx[X] [),] where InParameter *)
| {t=(TOPar _ | TComma _)}
::({t=TIdent (s, i1); where=InParameter::_} as tok1)
::{t=TOCro _}::_::{t=TCCro _}
::({t=(TCPar _ | TComma _)} as tok2)::xs ->
change_tok tok1 (TIdent_Typedef (s, i1));
aux (tok2::xs)
(* [(,] xx[...] could be a array access, so need InParameter guard *)
| {t=(TOPar _ | TComma _)}::({t=TIdent (s,i1);where=InParameter::_} as tok1)
::{t=TOCro _}::_tok::{t=TCCro _}::xs
->
change_tok tok1 (TIdent_Typedef (s, i1));
aux xs
(* kencc-ext: xx; where InStruct *)
| {t=tok_before}::({t=TIdent (s, i1)} as tok1)::({t=TPtVirg _} as tok2)::xs
when look_like_declaration_context tok_before ->
(match tok1.where with
| (InClassStruct _)::_ ->
change_tok tok1 (TIdent_Typedef (s, i1));
| _ -> ()
);
aux (tok2::xs)
(* sizeof(xx) sizeof expr does not require extra parenthesis, but
* in practice people do, so guard it with what looks_like_typedef
*)
| {t=Tsizeof _}::{t=TOPar _}::({t=TIdent (s, i1)} as tok1)::{t=TCPar _}::xs
when Token_views_context.look_like_typedef s || s =~ "^[A-Z].*" ->
change_tok tok1 (TIdent_Typedef (s, i1));
aux xs
(* new Xxx *)
| {t=Tnew _}::({t=TIdent (s, i1)} as tok1)::xs ->
change_tok tok1 (TIdent_Typedef (s, i1));
aux xs
(* recurse *)
| _::xs -> aux xs
in
xxs +> List.iter aux

View file

@ -0,0 +1,9 @@
val filter_for_typedef:
Token_views_cpp.multi_grouped list -> Token_views_cpp.token_extended list list
(* We use a list list because the template arguments are passed separately
* TODO: right now we actually skip template arguments ...
*)
val find_typedefs:
Token_views_cpp.token_extended list list -> unit

View file

@ -0,0 +1,140 @@
(* Yoann Padioleau
*
* Copyright (C) 2011 Facebook
* Copyright (C) 2006, 2007, 2008 Ecole des Mines de Nantes
*
* This program is free software; you can redistribute it and/or
* modify it under the terms of the GNU General Public License (GPL)
* version 2 as published by the Free Software Foundation.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* file license.txt for more details.
*)
open Common
module T = Parser_cpp
module TH = Token_helpers_cpp
module PI = Parse_info
(*****************************************************************************)
(* Wrappers *)
(*****************************************************************************)
let pr2_err, _pr2_once = Common2.mk_pr2_wrappers Flag_parsing_cpp.verbose_parsing
let pr2_err s = pr2_err ("ERROR_RECOV: " ^s)
(*****************************************************************************)
(* Helpers *)
(*****************************************************************************)
(*****************************************************************************)
(* Skipping stuff, find next "synchronisation" point *)
(*****************************************************************************)
(* todo: do something if find T.Eof ? *)
let rec find_next_synchro ~next ~already_passed =
(* Maybe because not enough }, because for example an ifdef contains
* in both branch some opening {, we later eat too much, "on deborde
* sur la fonction d'apres". So already_passed may be too big and
* looking for next synchro point starting from next may not be the
* best. So maybe we can find synchro point inside already_passed
* instead of looking in next.
*
* But take care! must progress. We must not stay in infinite loop!
* For instance now I have as a error recovery to look for
* a "start of something", corresponding to start of function,
* but must go beyond this start otherwise will loop.
* So look at premier(external_declaration2) in parser.output and
* pass at least those first tokens.
*
* I have chosen to start search for next synchro point after the
* first { I found, so quite sure we will not loop. *)
let last_round = List.rev already_passed in
let is_define =
let xs = last_round +> List.filter TH.is_not_comment in
match xs with
| T.TDefine _::_ -> true
| _ -> false
in
if is_define
then find_next_synchro_define (last_round @ next) []
else
let (before, after) =
last_round +> Common.span (fun tok ->
match tok with
(* by looking at TOBrace we are sure that the "start of something"
* will not arrive too early
*)
| T.TOBrace _ -> false
| T.TDefine _ -> false
| _ -> true
)
in
find_next_synchro_orig (after @ next) (List.rev before)
and find_next_synchro_define next already_passed =
match next with
| [] ->
pr2_err "end of file while in recovery mode";
already_passed, []
| (T.TCommentNewline_DefineEndOfMacro _ as v)::xs ->
pr2_err (spf "found sync end of #define at line %d" (TH.line_of_tok v));
v::already_passed, xs
| v::xs ->
find_next_synchro_define xs (v::already_passed)
and find_next_synchro_orig next already_passed =
match next with
| [] ->
pr2_err "end of file while in recovery mode";
already_passed, []
| (T.TCBrace i as v)::xs when PI.col_of_info i = 0 ->
pr2_err (spf "found sync '}' at line %d" (PI.line_of_info i));
(match xs with
| [] -> raise Impossible (* there is a EOF token normally *)
(* still useful: now parser.mly allow empty ';' so normally no pb *)
| T.TPtVirg iptvirg::xs ->
pr2_err "found sync bis, eating } and ;";
(T.TPtVirg iptvirg)::v::already_passed, xs
| T.TIdent x::T.TPtVirg iptvirg::xs ->
pr2_err "found sync bis, eating ident, }, and ;";
(T.TPtVirg iptvirg)::(T.TIdent x)::v::already_passed,
xs
| T.TCommentSpace sp::T.TIdent x::T.TPtVirg iptvirg
::xs ->
pr2_err "found sync bis, eating ident, }, and ;";
(T.TCommentSpace sp)::
(T.TPtVirg iptvirg)::
(T.TIdent x)::
v::
already_passed,
xs
| _ ->
v::already_passed, xs
)
| v::xs ->
let info = TH.info_of_tok v in
if PI.col_of_info info = 0 && TH.is_start_of_something v
then begin
pr2_err (spf "found sync col 0 at line %d " (PI.line_of_info info));
already_passed, v::xs
end
else find_next_synchro_orig xs (v::already_passed)

View file

@ -0,0 +1,5 @@
val find_next_synchro:
next:Parser_cpp.token list ->
already_passed:Parser_cpp.token list ->
Parser_cpp.token list * Parser_cpp.token list

View file

@ -0,0 +1,235 @@
(* Yoann Padioleau
*
* Copyright (C) 2007, 2008 Ecole des Mines de Nantes
* Copyright (C) 2011 Facebook
*
* This program is free software; you can redistribute it and/or
* modify it under the terms of the GNU General Public License (GPL)
* version 2 as published by the Free Software Foundation.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* file license.txt for more details.
*)
open Common
module Ast = Ast_cpp
module Flag = Flag_parsing_cpp
module TH = Token_helpers_cpp
module Parser = Parser_cpp
module Hack = Parsing_hacks_lib
open Parser_cpp
open Token_views_cpp
(*****************************************************************************)
(* Prelude *)
(*****************************************************************************)
(*
* CPP functions working at the token level. See pp_ast.ml for cpp functions
* working at the AST level (which is very unusual but makes sense in
* the coccinelle context for instance).
*
* Note that because I use a single lexer to work both at the C and cpp level
* there are some inconveniencies.
* For instance 'for' is a valid name for a macro parameter and macro
* body, but is interpreted in a special way by our single lexer, and
* so at some places where I expect a TIdent I need also to
* handle special cases and accept Tfor, Tif, etc at those places.
*
* There are multiple issues related to those keywords incorrect tokens.
* Those keywords can be:
*
* - (1) in the name of the macro as in #define inline
* - (2) in a parameter of the macro as in #define foo(char) char x;
* - (3) in an argument to a macro call as in IDENT(if);
*
* Case 1 is easy to fix in define_ident in ???
*
* Case 2 is easy to fix in define_parse below, where we detect such tokens
* in the parameters and then replace their occurence in the body with
* a TIdent.
*
* Case 3 is only an issue when the expanded token is not really used
* as usual but used for instance in concatenation as in a ## if
* when expanded. In the case the grammar this time will not be happy
* so this is also easy to fix in cpp_engine.
*)
(*****************************************************************************)
(* Wrappers *)
(*****************************************************************************)
let pr2, _pr2_once = Common2.mk_pr2_wrappers Flag_parsing_cpp.verbose_parsing
(*****************************************************************************)
(* Types *)
(*****************************************************************************)
(* the tokens in the body of the macro are all ExpandedTok *)
type define_body = (unit,string list) either * Parser_cpp.token list
(* TODO:
type define_def = string * define_param * define_body
and define_param =
| NoParam
| Params of string list
and define_body =
| DefineBody of Parser_c.token list
| DefineHint of parsinghack_hint
and parsinghack_hint =
| HintIterator
| HintDeclarator
| HintMacroString
| HintMacroStatement
| HintAttribute
| HintMacroIdentBuilder
*)
(*****************************************************************************)
(* Apply macro (using standard.h or other defs) *)
(*****************************************************************************)
(* cpp-builtin part1, macro, using standard.h or other defs *)
(* Thanks to this function many stuff are not anymore hardcoded in
* OCaml code (but are now hardcoded in standard.h ...)
*)
let (cpp_engine:
(string , Parser.token list) assoc -> Parser.token list -> Parser.token list)
= fun env xs ->
xs +> List.map (fun tok ->
match tok with
| TIdent (s,_i1) when List.mem_assoc s env -> Common2.assoc s env
| x -> [x]
)
+> List.flatten
(*
* We apply a macro by generating new ExpandedToken and by
* commenting the old macro call.
*
* no need to take care to substitute the macro name itself
* that occurs in the macro definition because the macro name is
* after fix_token_define a TDefineIdent, no more a TIdent.
*)
let apply_macro_defs defs xs =
let rec apply_macro_defs xs =
match xs with
| [] -> ()
(* recognized macro of standard.h (or other) *)
| PToken ({t=TIdent (s,_i1);_} as id)::Parenthised (xxs,info_parens)::xs
when Hashtbl.mem defs s ->
Hack.pr2_pp ("MACRO: found known macro = " ^ s);
(match Hashtbl.find defs s with
| Left (), bodymacro ->
pr2 ("macro without param used before parenthize, wierd: " ^ s);
(* ex: PRINTP("NCR53C400 card%s detected\n" ANDP(((struct ... *)
Hack.set_as_comment (Token_cpp.CppMacroExpanded) id;
id.new_tokens_before <- bodymacro;
| Right params, bodymacro ->
if List.length params = List.length xxs
then
let xxs' = xxs +> List.map (fun x ->
(tokens_of_paren_ordered x) +> List.map (fun x ->
TH.visitor_info_of_tok Ast.make_expanded x.t
)
) in
id.new_tokens_before <-
cpp_engine (Common2.zip params xxs') bodymacro
else begin
pr2 ("macro with wrong number of arguments, wierd: " ^ s);
id.new_tokens_before <- bodymacro;
end;
(* important to do that after have apply the macro, otherwise
* will pass as argument to the macro some tokens that
* are all TCommentCpp
*)
[Parenthised (xxs, info_parens)] +>
iter_token_paren (Hack.set_as_comment Token_cpp.CppMacroExpanded);
Hack.set_as_comment Token_cpp.CppMacroExpanded id;
);
apply_macro_defs xs
| PToken ({t=TIdent (s,_i1);_} as id)::xs
when Hashtbl.mem defs s ->
Hack.pr2_pp ("MACRO: found known macro = " ^ s);
(match Hashtbl.find defs s with
| Right _params, _bodymacro ->
pr2 ("macro with params but no parens found, wierd: " ^ s);
(* dont apply the macro, perhaps a redefinition *)
()
| Left (), bodymacro ->
(* special case when 1-1 substitution, we reuse the token *)
(match bodymacro with
| [newtok] ->
id.t <- (newtok +> TH.visitor_info_of_tok (fun _ ->
TH.info_of_tok id.t))
| _ ->
Hack.set_as_comment Token_cpp.CppMacroExpanded id;
id.new_tokens_before <- bodymacro;
)
);
apply_macro_defs xs
(* recurse *)
| (PToken _x)::xs -> apply_macro_defs xs
| (Parenthised (xxs, _info_parens))::xs ->
xxs +> List.iter apply_macro_defs;
apply_macro_defs xs
in
apply_macro_defs xs
(*****************************************************************************)
(* Extracting macros (from a standard.h) *)
(*****************************************************************************)
(* assumes have called fix_tokens_define before, so have TOPar_Define *)
let rec define_parse xs =
match xs with
| [] -> []
| TDefine _i1::TIdent_Define (s,_i2)::TOPar_Define _i3::xs ->
let (tokparams, _, xs) =
xs +> Common2.split_when (function TCPar _ -> true | _ -> false) in
let (body, _, xs) =
xs +> Common2.split_when
(function TCommentNewline_DefineEndOfMacro _ -> true | _ -> false) in
let params =
tokparams +> Common.map_filter (function
| TComma _ -> None
| TIdent (s, _) -> Some s
| x -> Common2.error_cant_have x
) in
let body = body +> List.map
(TH.visitor_info_of_tok Ast.make_expanded) in
let def = (s, (Right params, body)) in
def::define_parse xs
| TDefine _i1::TIdent_Define (s,_i2)::xs ->
let (body, _, xs) =
xs +> Common2.split_when
(function TCommentNewline_DefineEndOfMacro _ -> true | _ -> false) in
let body = body +> List.map
(TH.visitor_info_of_tok Ast.make_expanded) in
let def = (s, (Left (), body)) in
def::define_parse xs
| TDefine _i1::_ ->
raise Impossible
| _x::xs -> define_parse xs
let extract_macros xs =
let cleaner = xs +> List.filter (fun x -> not (TH.is_comment x)) in
define_parse cleaner

View file

@ -0,0 +1,59 @@
(* Expanding or extracting macros, at the token level *)
(* the either is to differentialte macro-variables from macro-functions *)
type define_body = (unit,string list) Common.either * Parser_cpp.token list
(* TODO
(* corresponds to what is in the yacfe configuration file (e.g. standard.h) *)
type define_def = string * define_param * define_body
and define_param =
| NoParam
| Params of string list
and define_body =
| DefineBody of Parser_c.token list
| DefineHint of parsinghack_hint
(* strongly corresponds to the TMacroXxx in the grammar and lexer and the
* MacroXxx in the ast.
*)
and parsinghack_hint =
| HintIterator
| HintDeclarator
| HintMacroString
| HintMacroStatement
| HintAttribute
| HintMacroIdentBuilder
*)
(* extracting define_def, e.g. from a standard.h; assume have called
* fix_tokens_define before to have the TDefEol *)
val extract_macros:
Parser_cpp.token list -> (string, define_body) Common.assoc
(* TODO
val string_of_define_def: define_def -> string
*)
(* used internally *)
(* This function work by side effect and may generate new tokens
* in the new_tokens_before field of the token_extended in the
* paren_grouped list. So don't forget to recall
* Token_views_c.rebuild_tokens_extented after this call, as well
* as probably insert_virtual_positions as new tokens
* are generated.
*
* note: it does not do some fixpoint, so the generated code may also
* contain some macros names.
*)
val apply_macro_defs:
(*
msg_apply_known_macro:(string -> unit) ->
msg_apply_known_macro_hint:(string -> unit) ->
?evaluate_concatop:bool ->
?inplace_when_single:bool ->
*)
(string, define_body (* define_def *)) Hashtbl.t ->
Token_views_cpp.paren_grouped list -> unit

View file

@ -0,0 +1,871 @@
open Common
open Parse_info
open Ast_cpp
module Flag = Flag_parsing_cpp
let process_either _of_a _of_b =
function
| Left left -> "" ^ _of_a left
| Right right -> "" ^ _of_b right
let process_option ofa x =
match x with
| None -> ""
| Some stuff -> "" ^ ofa stuff
let process_list _of_a node =
let map = List.map _of_a node
in String.concat ", " map
let rec process_info token =
process_token token
and process_token tok =
match tok.token with
| OriginTok loc -> loc.str
| FakeTokStr (v1, opt) -> ""
| Ab -> ""
| ExpandedTok (tok1, tok2, integer) -> tok1.str
and wrap _of_a (v1, v2) =
_of_a v1
and wrap2 _of_a (v1, v2) =
let v1 = _of_a v1 and v2 = process_info v2 in
v1 ^ v2
and process_paren _of_a (paren1, arglist, paren2) =
let paren1 = process_token paren1
and arglist = _of_a arglist
and paren2 = process_token paren2
in paren1 ^ arglist ^ paren2
and process_brace _of_a (br1, arglist, br2) =
_of_a arglist
and process_bracket _of_a (br1, arglist, br2) =
let br1 = process_token br1
and arglist = _of_a arglist
and br2 = process_token br2 in
br1 ^ arglist ^ br2
and process_angle _of_a (ang1, args, ang2) =
let ang1 = process_token ang1
and args = _of_a args
and ang2 = process_token ang2
in ang1 ^ args ^ ang2
and process_comma_list _of_a node =
process_list (wrap _of_a) node
and process_comma_list2 _of_a =
process_list (process_either _of_a process_token)
let rec process_token tok =
match tok.token with
| OriginTok loc -> loc.str
| FakeTokStr (v1, opt) -> ""
| Ab -> ""
| ExpandedTok (tok1, tok2, integer) -> tok1.str
and process_include_kind = function
| Local -> ""
| Standard -> ""
| Weird -> ""
and process_define_expr expr =
""
and process_constant =
function
| String (str, is_wchar) -> str
| MultiString -> ""
| Char (str, is_wchar) -> str
| Int str -> str
| Float (str, ftype) -> str
| Bool bval -> string_of_bool bval
and process_ident ident =
match ident with
| IdIdent (name, tok) ->
process_token tok
| IdTemplateId (ident, args) ->
""
| IdDestructor (tok, simple_ident) ->
let (_, idtok) = simple_ident in
"destructor" ^ process_token idtok
| IdOperator (tok, operator) ->
""
| IdConverter (tok, fullType) ->
""
and process_argument arg =
process_either process_expression process_weird_arg arg
and process_weird_arg =
function
| ArgType arg_type -> process_fullType arg_type
| ArgAction arg_action -> process_action_macro arg_action
and process_action_macro =
function
| ActMisc act_misc ->
process_list process_token act_misc
and process_typeC (tc, tok_list) =
process_typeCbis tc
and process_floatType =
function
| CFloat -> "cfloat"
| CDouble -> "cdouble"
| CLongDouble -> "clongdouble"
and process_intType =
function
| CChar -> "cchar"
| Si signed -> process_signed signed
| CBool -> "cbool"
| WChar_t -> "cwchar_t"
and process_signed (sign, base) =
let sign = process_sign sign and base = process_base base
in "c" ^ sign ^ base (* cuchar, cint, cuint, etc. *)
and process_base =
function
| CChar2 -> "char"
| CShort -> "short"
| CInt -> "int"
| CLong -> "long"
| CLongLong -> "longlong"
and process_sign =
function
| Signed -> ""
| UnSigned -> "u"
and process_baseType =
function
| Void -> "void"
| IntType intType -> process_intType intType
| FloatType floatType -> process_floatType floatType
and process_param_name =
function
| None -> ""
| Some (name, tok) -> process_token tok ^ ": "
and process_parameter {
p_name = p_name;
p_type = p_type;
p_register = p_register;
p_val = p_val
} =
let type_str = process_fullType p_type
and p_name = process_param_name p_name in
p_name ^ type_str
and process_functionType {
ft_ret = ft_ret;
ft_params = ft_params;
ft_dots = ft_dots;
ft_const = ft_const;
ft_throw = ft_throw
} =
let ret_type = process_fullType ft_ret
and paren_str =
process_paren (process_comma_list process_parameter) ft_params in
paren_str ^ ": " ^ ret_type
and process_simple_ident (name, tok) =
process_token tok
and process_e_val (tok, cexpr) =
let equals = process_token tok (* equals sign *)
and cexpr = process_constExpression cexpr (* const expr *)
in " " ^ equals ^ " " ^ cexpr
and process_enum_elem { e_name = e_name; e_val = e_val } =
let e_name = process_simple_ident e_name
and e_val = process_option process_e_val e_val in
e_name ^ e_val
and process_constExpression expr = process_expression expr
and process_template_arguments args =
process_angle (process_comma_list process_template_argument) args
and process_template_argument arg =
process_either process_fullType process_expression arg
and process_qualifier =
function
| QClassname ((name, info)) ->
name ^ process_info info
| QTemplateId ((name, args)) ->
let ident = process_simple_ident name
and args = process_template_arguments args in
ident ^ args
and process_name (v1, v2, v3) =
let v1 = process_option process_token v1
and v2 =
process_list
(fun (v1, v2) ->
let v1 = process_qualifier v1
and v2 = process_token v2 in
v1 ^ v2)
v2
and v3 = process_ident v3
in v1 ^ v2 ^ v3
and process_either_ft_or_expr ft_or_expr =
process_either process_fullType process_expression ft_or_expr
and process_structUnion =
function
| Struct -> "struct"
| Union -> "union"
| Class -> "class"
and process_typeCbis =
function
| BaseType btype ->
process_baseType btype
| Pointer point ->
"ptr " ^ process_fullType point
| Reference ref ->
"ref " ^ process_fullType ref
| Array ((arr, typ)) ->
let arr = process_bracket (process_option process_constExpression) arr
and typ = process_fullType typ
in arr ^ typ
| FunctionType ftype ->
process_functionType ftype
| EnumDef ((name, ident, elements)) ->
let ident = process_option process_simple_ident ident
and elements =
process_brace (process_comma_list process_enum_elem) elements
in ident ^ " = enum\n" ^ elements
| StructDef sdef ->
"" (*process_class_definition sdef*)
| EnumName ((enum, name)) ->
process_simple_ident name
| StructUnionName ((stype_tok, name)) ->
let (stype, _) = stype_tok in
let stype = process_structUnion stype
and name = process_simple_ident name in
stype ^ " " ^ name
| TypeName ((tname)) ->
process_name tname
| TypenameKwd ((tname (* 'typename' *), tdef_name)) ->
process_name tdef_name
| TypeOf ((typeof, tdef)) ->
process_paren process_either_ft_or_expr tdef
| ParenType paren ->
process_paren process_fullType paren
and process_info token =
process_token token
and process_expression (expr, toks) =
process_exprbis expr
and process_exprbis =
function
| Id ((name, info)) ->
let (_, _, ident) = name in
process_ident ident
| C const -> process_constant const
| Call ((expr, args)) ->
let name = process_expression expr
and args = process_paren (process_comma_list process_argument) args
in name ^ args
| CondExpr ((v1, v2, v3)) ->
(*let v1 = vof_expression v1
and v2 = Ocaml.vof_option vof_expression v2
and v3 = vof_expression v3*)
""
| Sequence ((v1, v2)) ->
(*let v1 = vof_expression v1
and v2 = vof_expression v2
in Ocaml.VSum (("Sequence", [ v1; v2 ]))*)
""
| Assignment ((v1, v2, v3)) ->
(*let v1 = vof_expression v1
and v2 = vof_assignOp v2
and v3 = vof_expression v3
in Ocaml.VSum (("Assignment", [ v1; v2; v3 ]))*)
""
| Postfix ((v1, v2)) ->
(*let v1 = vof_expression v1
and v2 = vof_fixOp v2
in Ocaml.VSum (("Postfix", [ v1; v2 ]))*)
""
| Infix ((v1, v2)) ->
(*let v1 = vof_expression v1
and v2 = vof_fixOp v2
in Ocaml.VSum (("Infix", [ v1; v2 ]))*)
""
| Unary ((v1, v2)) ->
(*let v1 = vof_expression v1
and v2 = vof_unaryOp v2
in Ocaml.VSum (("Unary", [ v1; v2 ]))*)
""
| Binary ((v1, v2, v3)) ->
(*let v1 = vof_expression v1
and v2 = vof_binaryOp v2
and v3 = vof_expression v3
in Ocaml.VSum (("Binary", [ v1; v2; v3 ]))*)
""
| ArrayAccess ((v1, v2)) ->
(*let v1 = vof_expression v1
and v2 = vof_bracket vof_expression v2
in Ocaml.VSum (("ArrayAccess", [ v1; v2 ]))*)
""
| RecordAccess ((v1, v2)) ->
(*let v1 = vof_expression v1
and v2 = vof_name v2
in Ocaml.VSum (("RecordAccess", [ v1; v2 ]))*)
""
| RecordPtAccess ((v1, v2)) ->
(*let v1 = vof_expression v1
and v2 = vof_name v2
in Ocaml.VSum (("RecordPtAccess", [ v1; v2 ]))*)
""
| RecordStarAccess ((v1, v2)) ->
(*let v1 = vof_expression v1
and v2 = vof_expression v2
in Ocaml.VSum (("RecordStarAccess", [ v1; v2 ]))*)
""
| RecordPtStarAccess ((v1, v2)) ->
(*let v1 = vof_expression v1
and v2 = vof_expression v2
in Ocaml.VSum (("RecordPtStarAccess", [ v1; v2 ]))*)
""
| SizeOfExpr ((v1, v2)) ->
(*let v1 = vof_tok v1
and v2 = vof_expression v2
in Ocaml.VSum (("SizeOfExpr", [ v1; v2 ]))*)
""
| SizeOfType ((v1, v2)) ->
(*let v1 = vof_tok v1
and v2 = vof_paren vof_fullType v2
in Ocaml.VSum (("SizeOfType", [ v1; v2 ]))*)
""
| Cast ((v1, v2)) ->
(*let v1 = vof_paren vof_fullType v1
and v2 = vof_expression v2
in Ocaml.VSum (("Cast", [ v1; v2 ]))*)
""
| StatementExpr v1 ->
(*let v1 = vof_paren vof_compound v1
in Ocaml.VSum (("StatementExpr", [ v1 ]))*)
""
| GccConstructor ((v1, v2)) ->
(*let v1 = vof_paren vof_fullType v1
and v2 = vof_brace (vof_comma_list vof_initialiser) v2
in Ocaml.VSum (("GccConstructor", [ v1; v2 ]))*)
""
| This v1 ->
(*let v1 = vof_tok v1 in Ocaml.VSum (("This", [ v1 ]))*)
""
| ConstructedObject ((v1, v2)) ->
(*let v1 = vof_fullType v1
and v2 = vof_paren (vof_comma_list vof_argument) v2
in Ocaml.VSum (("ConstructedObject", [ v1; v2 ]))*)
""
| TypeId ((v1, v2)) ->
(*let v1 = vof_tok v1
and v2 = vof_paren vof_either_ft_or_expr v2
in Ocaml.VSum (("TypeId", [ v1; v2 ]))*)
""
| CplusplusCast ((v1, v2, v3)) ->
(*let v1 = vof_wrap2 vof_cast_operator v1
and v2 = vof_angle vof_fullType v2
and v3 = vof_paren vof_expression v3
in Ocaml.VSum (("CplusplusCast", [ v1; v2; v3 ]))*)
""
| New ((v1, v2, v3, v4, v5)) ->
(*let v1 = Ocaml.vof_option vof_tok v1
and v2 = vof_tok v2
and v3 = Ocaml.vof_option (vof_paren (vof_comma_list vof_argument)) v3
and v4 = vof_fullType v4
and v5 = Ocaml.vof_option (vof_paren (vof_comma_list vof_argument)) v5
in Ocaml.VSum (("New", [ v1; v2; v3; v4; v5 ]))*)
""
| Delete ((v1, v2)) ->
(*let v1 = Ocaml.vof_option vof_tok v1
and v2 = vof_expression v2
in Ocaml.VSum (("Delete", [ v1; v2 ]))*)
""
| DeleteArray ((tok, expr)) ->
let tok = process_option process_token tok
and expr = process_expression expr
in tok ^ expr
| Throw throw ->
process_option process_expression throw
| ParenExpr paren_expr ->
process_paren process_expression paren_expr
| ExprTodo -> "TODO"
and process_selection =
function
| If ((v1, v2, v3, v4, v5)) ->
let v1 = process_token v1
and v2 = process_paren process_expression v2
and v3 = process_statement v3
and v4 = process_option process_token v4
and v5 = process_statement v5
in v1 ^ v2 ^ v3 ^ v4 ^v5
| Switch ((v1, v2, v3)) ->
let v1 = process_token v1
and v2 = process_paren process_expression v2
and v3 = process_statement v3
in v1 ^ v2 ^ v3
and process_iteration =
function
| While ((v1, v2, v3)) ->
let v1 = process_token v1
and v2 = process_paren process_expression v2
and v3 = process_statement v3
in v1 ^ v2 ^ v3
| DoWhile ((v1, v2, v3, v4, v5)) ->
let v1 = process_token v1
and v2 = process_statement v2
and v3 = process_token v3
and v4 = process_paren process_expression v4
and v5 = process_token v5
in v1 ^ v2 ^ v3 ^ v4 ^v5
| For ((v1, v2, v3)) ->
let v1 = process_token v1
and v2 =
process_paren
(fun (v1, v2, v3) ->
let v1 = wrap process_exprStatement v1
and v2 = wrap process_exprStatement v2
and v3 = wrap process_exprStatement v3
in v1 ^ v2 ^ v3)
v2
and v3 = process_statement v3
in v1 ^ v2 ^ v3
| MacroIteration ((v1, v2, v3)) ->
let v1 = process_simple_ident v1
and v2 = process_paren (process_comma_list process_argument) v2
and v3 = process_statement v3
in v1 ^ v2 ^ v3
and process_jump =
function
| Goto goto -> "# XXX goto not supported: " ^ goto
| Continue -> "continue"
| Break -> "break"
| Return -> "return"
| ReturnExpr ret_expr ->
process_expression ret_expr
| GotoComputed goto_comp ->
"#[ XXX goto not supported: " ^ process_expression goto_comp ^ "]#"
and process_handler (v1, v2, v3) =
let v1 = process_token v1
and v2 = process_paren process_exception_declaration v2
and v3 = process_compound v3
in v1 ^ v2 ^ v3
and process_exception_declaration =
function
| ExnDeclEllipsis exn_ellipsis ->
process_token exn_ellipsis
| ExnDecl exn_decl ->
process_parameter exn_decl
and get_tydef_prefix name storage =
match storage with
| NoSto -> ""
| StoTypedef st_tdef ->
"type " ^ name ^ " = "
| Sto (sto, tok) -> ""
and process_onedecl {
v_namei = v_namei;
v_type = v_type;
v_storage = v_storage
} =
let name =
process_option
(fun (name, init) ->
let name = process_name name
and init = process_option process_init init
in name ^ init)
v_namei in
let res = process_onedeclFullType "" name v_storage v_type in
res
and process_onedeclFullType prefix name storage (qualifier, (typeCbis, tok_list)) =
match typeCbis with
| BaseType btype ->
process_baseType btype
| Pointer point ->
process_onedeclFullType "ptr " name storage point
| Reference ref ->
process_onedeclFullType "ref " name storage ref
| Array ((arr, typ)) ->
let arr = process_bracket (process_option process_constExpression) arr
and typ = process_fullType typ
in arr ^ typ
| FunctionType ftype ->
let ret = match storage with
| NoSto -> "proc " ^ name ^ process_functionType ftype
| StoTypedef st_tdef ->
"type " ^ name ^ " = " ^ "proc " ^ process_functionType ftype
| Sto sto -> "proc " ^ name ^ process_functionType ftype in
ret
| EnumDef ((name, ident, elements)) ->
let ident = process_option process_simple_ident ident
and elements =
process_brace (process_comma_list process_enum_elem) elements
in "type " ^ ident ^ " = enum " ^ elements
| StructDef sdef ->
"" (*process_class_definition sdef*)
| EnumName ((enum, name)) ->
process_simple_ident name
| StructUnionName ((stype_tok, name)) ->
let (stype, _) = stype_tok in
let stype = process_structUnion stype
and name = process_simple_ident name in
stype ^ " " ^ name
| TypeName ((tname)) ->
process_name tname
| TypenameKwd ((tname (* 'typename' *), tdef_name)) ->
process_name tdef_name
| TypeOf ((typeof, tdef)) ->
process_paren process_either_ft_or_expr tdef
| ParenType (left, type_inf, right) ->
process_onedeclFullType "" name storage type_inf
and process_storage st = process_storagebis st
and process_storagebis =
function
| NoSto -> ""
| StoTypedef st_tdef ->
process_token st_tdef
| Sto sto -> wrap2 process_storageClass sto
and process_storageClass =
function
| Auto -> "auto"
| Static -> "static"
| Register -> "register"
| Extern -> "extern"
and process_init =
function
| EqInit ((v1, v2)) ->
let v1 = process_token v1
and v2 = process_initialiser v2
in v1 ^ v2
| ObjInit v1 ->
process_paren (process_comma_list process_argument) v1
and process_block_declaration =
function
| DeclList ((decl, semi_col)) ->
let v1 = process_comma_list process_onedecl decl
in "DECLLIST " ^ v1
| MacroDecl ((v1, v2, v3, v4)) ->
let v1 = process_list process_token v1
and v2 = process_simple_ident v2
and v3 = process_paren (process_comma_list process_argument) v3
and v4 = process_token v4
in v1 ^ v2 ^ v3 ^ v4
| UsingDecl v1 ->
let v1 =
(match v1 with
| (v1, v2, v3) ->
let v1 = process_token v1
and v2 = process_name v2
and v3 = process_token v3
in v1 ^ v2 ^ v3)
in v1
| UsingDirective ((v1, v2, v3, v4)) ->
let v1 = process_token v1
and v2 = process_token v2
and v3 = process_name v3
and v4 = process_token v4
in v1 ^ v2 ^ v3 ^ v4
| NameSpaceAlias ((v1, v2, v3, v4, v5)) ->
let v1 = process_token v1
and v2 = process_simple_ident v2
and v3 = process_token v3
and v4 = process_name v4
and v5 = process_token v5
in v1 ^ v2 ^ v3 ^ v4 ^ v5
| Asm ((v1, v2, v3, v4)) ->
let v1 = process_token v1
and v2 = process_option process_token v2
and v3 = process_paren process_asmbody v3
and v4 = process_token v4
in v1 ^ v2 ^ v3 ^ v4
and process_asmbody (v1, v2) =
let v1 = process_list process_token v1
and v2 = process_list (wrap process_colon) v2
in v1 ^ v2
and process_colon =
function
| Colon v1 ->
let v1 = process_comma_list process_colon_option v1
in v1
and process_colon_option v = wrap process_colon_optionbis v
and process_colon_optionbis =
function
| ColonMisc -> "colonmisc"
| ColonExpr v1 ->
let v1 = process_paren process_expression v1
in v1
and process_statement stmt = wrap process_statementbis stmt
and process_statementbis =
function
| Compound comp ->
process_compound comp
| ExprStatement expr ->
process_exprStatement expr
| Labeled labeled ->
process_labeled labeled
| Selection selection ->
process_selection selection
| Iteration iter ->
process_iteration iter
| Jump jump ->
process_jump jump
| DeclStmt decl ->
process_block_declaration decl
| Try ((tok, comp, handler_list)) ->
let comp = process_compound comp
and handler_list = process_list process_handler handler_list
in "try: " ^ comp ^ handler_list
| NestedFunc nest_func ->
process_func_definition nest_func
| MacroStmt -> ""
| StmtTodo -> "# TODO"
and process_compound comp = process_brace (process_list process_statement_sequencable) comp
and process_statement_sequencable =
function
| StmtElem stmt ->
process_statement stmt
| CppDirectiveStmt direc ->
process_cpp_directive direc
| IfdefStmt ifdef ->
process_ifdef_directive ifdef
and process_ifdef_directive if_def = wrap2 process_ifdefkind if_def
and process_ifdefkind =
function (* TODO fix this for Nim *)
| Ifdef -> "ifdef"
| IfdefElse -> "ifdefelse"
| IfdefElseif -> "ifdefelseif"
| IfdefEndif -> "ifdefendif"
and process_exprStatement expr_stmt =
process_option process_expression expr_stmt
and process_labeled =
function
| Label ((name, stmt)) ->
let stmt = process_statement stmt
in name ^ " " ^ stmt
| Case ((expr, stmt)) ->
let expr = process_expression expr
and stmt = process_statement stmt
in expr ^ " " ^stmt
| CaseRange ((expr1, expr2, stmt)) ->
let expr1 = process_expression expr1
and expr2 = process_expression expr2
and stmt = process_statement stmt
in expr1 ^ expr2 ^ stmt
| Default def ->
process_statement def
and process_initialiser =
function
| InitExpr v1 ->
process_expression v1
| InitList v1 ->
process_brace (process_comma_list process_initialiser) v1
| InitDesignators ((v1, v2, v3)) ->
let v1 = process_list process_designator v1
and v2 = process_token v2
and v3 = process_initialiser v3
in v1 ^ v2 ^ v3
| InitFieldOld ((v1, v2, v3)) ->
let v1 = process_simple_ident v1
and v2 = process_token v2
and v3 = process_initialiser v3
in v1 ^ v2 ^ v3
| InitIndexOld ((v1, v2)) ->
let v1 = process_bracket process_expression v1
and v2 = process_initialiser v2
in v1 ^ v2
and process_designator =
function
| DesignatorField ((v1, v2)) ->
let v1 = process_token v1
and v2 = process_simple_ident v2
in v1 ^ v2
| DesignatorIndex v1 ->
process_bracket process_expression v1
| DesignatorRange v1 ->
process_bracket
(fun (v1, v2, v3) ->
let v1 = process_expression v1
and v2 = process_token v2
and v3 = process_expression v3
in v1 ^ v2 ^ v3)
v1
and process_define_val =
function
| DefinePrintWrapper ((if_tok, expr_paren, name)) ->
let expr_paren = process_paren process_expression expr_paren
and name = process_name name in
expr_paren ^ name
| DefineExpr expr ->
process_expression expr
| DefineStmt stmt ->
process_statement stmt
| DefineType dtype ->
process_fullType dtype
| DefineDoWhileZero (stmt, tok_list) ->
process_statement stmt
| DefineFunction dfunc ->
process_func_definition dfunc
| DefineInit init ->
process_initialiser init
| DefineText (str, toks) ->
str
| DefineEmpty -> ""
| DefineTodo -> ""
and process_define _tok ident kind value =
match kind with
| DefineVar ->
let (idname, _ ) = ident in
"const " ^ idname ^ " = " ^ process_define_val value ^ "\n"
| DefineFunc func ->
""
(*let (idname, _) = ident
in *)
and process_include ((tok, kind, path)) =
let include_file =
match kind with
| Local -> path
| Standard -> path
| Weird ->
let search = Str.regexp "_"
and lower = String.lowercase_ascii path
in Str.global_replace search "." lower
in "#" ^ include_file ^ " " ^ process_token tok
and process_cpp_directive = function
| Define ((tok, ident, kind, value)) ->
process_define tok ident kind value
| Include ((tok, inc_kind, path)) ->
process_include (tok, inc_kind, path)
| Undef ((name, tok)) ->
process_token tok
| PragmaAndCo tok ->
process_token tok
and process_func_definition {
f_name = f_name;
f_type = f_type;
f_storage = f_storage;
f_body = f_body
} =
""
and process_func_or_else =
function
| FunctionOrMethod func_meth ->
process_func_definition func_meth
| Constructor ((func)) ->
process_func_definition func
| Destructor func ->
process_func_definition func
and process_declaration =
function
| BlockDecl block ->
process_block_declaration block
| Func func ->
(*let v1 = vof_func_or_else v1 in Ocaml.VSum (("Func", [ v1 ]))*)
process_func_or_else func
| TemplateDecl (v1, v2, v3) ->
(*let v1 = vof_tok v1
and v2 = vof_template_parameters v2
and v3 = vof_declaration v3
in Ocaml.VSum (("TemplateDecl", [ v1; v2; v3 ]))*)
""
| TemplateSpecialization ((v1, v2, v3)) ->
(*let v1 = vof_tok v1
and v2 = vof_angle Ocaml.vof_unit v2
and v3 = vof_declaration v3
in Ocaml.VSum (("TemplateSpecialization", [ v1; v2; v3 ]))*)
""
| ExternC ((v1, v2, v3)) ->
(*let v1 = vof_tok v1
and v2 = vof_tok v2
and v3 = vof_declaration v3
in Ocaml.VSum (("ExternC", [ v1; v2; v3 ]))*)
""
| ExternCList ((v1, v2, v3)) ->
(*let v1 = vof_tok v1
and v2 = vof_tok v2
and v3 = vof_brace (Ocaml.vof_list vof_declaration_sequencable) v3
in Ocaml.VSum (("ExternCList", [ v1; v2; v3 ]))*)
""
| NameSpace ((v1, v2, v3)) ->
(*let v1 = vof_tok v1
and v2 = vof_wrap2 Ocaml.vof_string v2
and v3 = vof_brace (Ocaml.vof_list vof_declaration_sequencable) v3
in Ocaml.VSum (("NameSpace", [ v1; v2; v3 ]))*)
""
| NameSpaceExtend ((v1, v2)) ->
(*let v1 = Ocaml.vof_string v1
and v2 = Ocaml.vof_list vof_declaration_sequencable v2
in Ocaml.VSum (("NameSpaceExtend", [ v1; v2 ]))*)
""
| NameSpaceAnon ((v1, v2)) ->
(*let v1 = vof_tok v1
and v2 = vof_brace (Ocaml.vof_list vof_declaration_sequencable) v2
in Ocaml.VSum (("NameSpaceAnon", [ v1; v2 ]))*)
""
| EmptyDef def -> process_token def
| DeclTodo -> "# TODO"
and process_fullType ((qualifier, typeC)) =
process_typeC typeC
and process_toplevel = function
| NotParsedCorrectly node -> ""
| DeclElem node -> process_declaration node
| CppDirectiveDecl node -> process_cpp_directive node
| IfdefDecl node -> ""
| MacroTop ((v1, v2, v3)) -> ""
| MacroVarTop ((v1, v2)) -> ""
let iter_ast ast =
List.map process_toplevel ast
let test_dump_nim file =
Parse_cpp.init_defs !Flag.macros_h;
let ast = Parse_cpp.parse_program file in
let res = iter_ast ast in
List.iter pr res

View file

@ -0,0 +1,2 @@
val test_dump_nim :
Common.filename -> unit

View file

@ -0,0 +1,115 @@
open Common
open Parse_info
open Ast_cpp
module Ast = Ast_cpp
module Flag = Flag_parsing_cpp
module TH = Token_helpers_cpp
module Stat = Parse_info
(*****************************************************************************)
(* Subsystem testing *)
(*****************************************************************************)
let test_tokens_cpp file =
Flag.verbose_lexing := true;
Flag.verbose_parsing := true;
let toks = Parse_cpp.tokens file in
toks +> List.iter (fun x -> pr2_gen x);
()
let test_dump_cpp file =
Parse_cpp.init_defs !Flag.macros_h;
let ast = Parse_cpp.parse_program file in
let v = Meta_ast_cpp.vof_program ast in
let s = Ocaml.string_of_v v in
pr s
let test_dump_cpp_full file =
Parse_cpp.init_defs !Flag.macros_h;
let ast = Parse_cpp.parse_program file in
let toks = Parse_cpp.tokens file in
let precision = { Meta_ast_generic.
full_info = true; type_info = true; token_info = true;
}
in
let v = Meta_ast_cpp.vof_program ~precision ast in
let s = Ocaml.string_of_v v in
pr s;
toks +> List.iter (fun tok ->
match tok with
| Parser_cpp.TComment (ii) ->
let v = Parse_info.vof_info ii in
let s = Ocaml.string_of_v v in
pr s
| _ -> ()
);
()
let test_dump_cpp_view file =
Parse_cpp.init_defs !Flag.macros_h;
let toks_orig = Parse_cpp.tokens file in
let toks =
toks_orig +> Common.exclude (fun x ->
Token_helpers_cpp.is_comment x ||
Token_helpers_cpp.is_eof x
)
in
let extended = toks +> List.map Token_views_cpp.mk_token_extended in
Parsing_hacks_cpp.find_template_inf_sup extended;
let multi = Token_views_cpp.mk_multi extended in
Token_views_context.set_context_tag_multi multi;
let v = Token_views_cpp.vof_multi_grouped_list multi in
let s = Ocaml.string_of_v v in
pr s
let test_parse_cpp_fuzzy xs =
let fullxs = Lib_parsing_cpp.find_source_files_of_dir_or_files xs
+> Skip_code.filter_files_if_skip_list
in
fullxs +> Console.progress (fun k -> List.iter (fun file ->
k ();
Common.save_excursion Flag_parsing_cpp.strict_lexer true (fun () ->
try
let _fuzzy = Parse_cpp.parse_fuzzy file in
()
with exn ->
pr2 (spf "PB with: %s, exn = %s" file (Common.exn_to_s exn));
)
))
let test_dump_cpp_fuzzy file =
let fuzzy, _toks = Parse_cpp.parse_fuzzy file in
let v = Ast_fuzzy.vof_trees fuzzy in
let s = Ocaml.string_of_v v in
pr2 s
(*****************************************************************************)
(* Main entry for Arg *)
(*****************************************************************************)
let actions () = [
"-tokens_cpp", " <file>",
Common.mk_action_1_arg test_tokens_cpp;
"-dump_cpp", " <file>",
Common.mk_action_1_arg test_dump_cpp;
"-dump_nim", " <file>",
Common.mk_action_1_arg Test_dump_nim.test_dump_nim;
"-dump_cpp_full", " <file>",
Common.mk_action_1_arg test_dump_cpp_full;
"-dump_cpp_view", " <file>",
Common.mk_action_1_arg test_dump_cpp_view;
"-parse_cpp_fuzzy", " <files or dirs>",
Common.mk_action_n_arg test_parse_cpp_fuzzy;
"-dump_cpp_fuzzy", " <file>",
Common.mk_action_1_arg test_dump_cpp_fuzzy;
]

View file

@ -0,0 +1,12 @@
(* Print the set of tokens in a c++ file *)
val test_tokens_cpp :
Common.filename -> unit
val test_dump_cpp:
Common.filename -> unit
(* This makes accessible the different test_xxx functions above from
* the command line, e.g. '$ pfff -parse_cpp foo.cpp will call the
* test_parse_cpp function.
*)
val actions : unit -> Common.cmdline_actions

View file

@ -0,0 +1,144 @@
was in token_views_context.ml:
(*
let look_like_only_idents xs =
xs +> List.for_all (function
| Tok {t=(TComma _ | TIdent _)} -> true
(* when have cast *)
| Parens _ -> true
| _ -> false
)
*)
(*
| BToken ({t=tokstruct; _})::BToken ({t= TIdent (s,_); _})
::Braceised(body, tok1, tok2)::xs when TH.is_classkey_keyword tokstruct ->
body +> List.iter (iter_token_brace (fun tok ->
tok.where <- (InClassStruct s)::tok.where;
));
set_in_other xs
(* struct/union/class x : ... { } *)
| BToken ({t= tokstruct; _})::BToken ({t=TIdent _; _})
::BToken ({t=TCol _})::xs when TH.is_classkey_keyword tokstruct ->
(try
let (before, elem, after) = Common2.split_when is_braceised xs in
(match elem with
| Braceised(body, tok1, tok2) ->
body +> List.iter (iter_token_brace (fun tok ->
tok.where <- InInitializer::tok.where;
));
set_in_other after
| _ -> raise Impossible
)
with Not_found ->
pr2 ("PB: could not find braces after struct/union/class x : ...");
)
*)
(* todo: this lead to some regressions :(
(* = ... ; *)
| Tok ({t=TEq ii;where = [InTopLevel]})::xs ->
let (before, ptvirg, after) =
try
xs +> Common2.split_when (function
| Tok ({t=TPtVirg _;}) -> true
| _ -> false
)
with Not_found ->
raise (UnclosedSymbol (spf "PB with split_when at %s"
(Parse_info.string_of_info ii)))
in
before +> TV.iter_token_multi (fun tok ->
tok.TV.where <- TV.InAssign::tok.TV.where;
);
aux before;
aux [ptvirg];
aux after
*)
(* TODO xx(...) { InFunction (can have some try or const or throw after
* the paren *)
(* could try: ) { } but it can be the ) of a if or while, so
* better to base the heuristic on the position in column zero.
* Note that some struct or enum or init put also their { in first column
* but set_in_other will overwrite the previous InFunction tag.
*)
(*TODOC++ext: now can have some const or throw between
=> do a view that filter them first ?
*)
(*
(* ) { and the closing } is in column zero, then certainly a function *)
(*TODO1 col 0 not valid anymore with c++ nestedness of method *)
| BToken ({t=TCPar _})::(Braceised (body, tok1, Some tok2))::xs
when tok1.col <> 0 && tok2.col = 0 ->
body +> List.iter (iter_token_brace (fun tok ->
tok.where <- InFunction::tok.where;
));
aux xs
| (BToken x)::xs -> aux xs
(*TODO1 not valid anymore with c++ nestedness of method *)
| (Braceised (body, tok1, Some tok2))::xs
when tok1.col = 0 && tok2.col = 0 ->
body +> List.iter (iter_token_brace (fun tok ->
tok.where <- InFunction::tok.where;
));
aux xs
| Braceised (body, tok1, tok2)::xs ->
aux xs
in
(* TODO <...> InTemplateParam *)
*)
(* C++: second tentative on InArgument, if xx(xx, yy, ww) where have only
* identifiers, it's probably a constructed object!
* But FP on C code, so should guard that with Flag_parsing_cpp.lang = C++
*)
(*
| Tok{t=TIdent _; where = ctx}::(Parens(_t1, body, _t2) as parens)::xs
when List.length body > 0 && look_like_only_idents body ->
[parens] +> TV.iter_token_multi (fun tok ->
let where =
match ctx with
| TV.InTopLevel::_ -> TV.InParameter
| TV.InAssign::_ -> TV.InArgument
| _ -> TV.InArgument
in
tok.TV.where <- where::tok.TV.where;
);
(* todo? recurse on body? *)
aux (parens::xs)
(* could be a cast too ... or what else? *)
| x::(Parens(_t1, _body, _t2) as parens)::xs ->
(* let's default to something? hmm, no, got lots of regressions then
* old: msg_context t1.t (TV.InArgument); ...
*)
aux [x];
aux (parens::xs)
*)

164
lang_cpp/parsing/todo_ml Normal file
View file

@ -0,0 +1,164 @@
let noInstr = (ExprStatement (None), [])
(*****************************************************************************)
(* Wrappers *)
(*****************************************************************************)
let unwrap_expr ((unwrap_e, typ), iie) = unwrap_e
let rewrap_expr ((_old_unwrap_e, typ), iie) newe = ((newe, typ), iie)
let get_type_expr ((unwrap_e, typ), iie) = !typ
let set_type_expr ((unwrap_e, oldtyp), iie) newtyp =
oldtyp := newtyp
(* old: (unwrap_e, newtyp), iie *)
let unwrap_typeC (qu, (typeC, ii)) = typeC
let rewrap_typeC (qu, (typeC, ii)) newtypeC = (qu, (newtypeC, ii))
let is_fake ii =
match ii.pinfo with
FakeTok (_,_) -> true
| _ -> false
let mcode_of_info ii = fst (!(ii.cocci_tag))
type posrv = Real of Common.parse_info | Virt of virtual_position
let compare_pos ii1 ii2 =
let get_pos = function
OriginTok pi -> Real pi
| FakeTok (s,vpi) -> Virt vpi
| ExpandedTok (pi,vpi) -> Virt vpi
| AbstractLineTok pi -> Real pi in (* used for printing *)
let pos1 = get_pos (pinfo_of_info ii1) in
let pos2 = get_pos (pinfo_of_info ii2) in
match (pos1,pos2) with
(Real p1, Real p2) -> compare p1.Common.charpos p2.Common.charpos
| (Virt (p1,_), Real p2) ->
if (compare p1.Common.charpos p2.Common.charpos) = (-1) then (-1) else 1
| (Real p1, Virt (p2,_)) ->
if (compare p1.Common.charpos p2.Common.charpos) = 1 then 1 else (-1)
| (Virt (p1,o1), Virt (p2,o2)) ->
let poi1 = p1.Common.charpos in
let poi2 = p2.Common.charpos in
match compare poi1 poi2 with
-1 -> -1
| 0 -> compare o1 o2
| x -> x
let equal_posl (l1,c1) (l2,c2) =
(l1 =|= l2) && (c1 =|= c2)
let info_to_fixpos ii =
match pinfo_of_info ii with
OriginTok pi -> Ast_cocci.Real pi.Common.charpos
| ExpandedTok (_,(pi,offset)) ->
Ast_cocci.Virt (pi.Common.charpos,offset)
| FakeTok (_,(pi,offset)) ->
Ast_cocci.Virt (pi.Common.charpos,offset)
| AbstractLineTok pi -> failwith "unexpected abstract"
(*****************************************************************************)
(* Abstract line *)
(*****************************************************************************)
(* When we have extended the C Ast to add some info to the tokens,
* such as its line number in the file, we can not use anymore the
* ocaml '=' to compare Ast elements. To overcome this problem, to be
* able to use again '=', we just have to get rid of all those extra
* information, to "abstract those line" (al) information.
*)
let al_info tokenindex x =
{ pinfo =
(AbstractLineTok
{charpos = tokenindex;
line = tokenindex;
column = tokenindex;
file = "";
str = str_of_info x});
cocci_tag = ref emptyAnnot;
comments_tag = ref emptyComments;
}
let semi_al_info x =
{ x with
cocci_tag = ref emptyAnnot;
comments_tag = ref emptyComments;
}
(*****************************************************************************)
(* Helpers *)
(*****************************************************************************)
(* todo? could also stringify the all ident? *)
let string_of_name name =
let (_qtop, _scope, (ident, _ii)) = name in
match ident with
| IdIdent s -> s
| IdOperator op -> "op todo"
| IdConverter ft -> "converter todo"
| IdDestructor xx -> "destructor todo"
| IdTemplateId (s, args) -> "template todo"
let is_simple_ident name =
let (qtop, scope, (ident, _ii)) = name in
match qtop, scope, ident with
| None, [], IdIdent _ -> true
| _ -> false
(* good to look at su ? some people use class for struct too ? *)
let is_class_structunion class_def =
let (su, _sopt, _bopt, (members: class_member_sequencable list)),_ii = class_def in
su = Class ||
(members +> List.exists (fun x ->
match x with
| ClassElem (DeclarationField _,ii) -> false
| ClassElem (EmptyField, ii) -> false
| _ -> true
))
(*****************************************************************************)
(* Views *)
(*****************************************************************************)
(* Transform a list of arguments (or parameters) where the commas are
* represented via the wrap2 and associated with an element, with
* a list where the comma are on their own. f(1,2,2) was
* [(1,[]); (2,[,]); (2,[,])] and become [1;',';2;',';2].
*
* Used in cocci_vs_c.ml, to have a more direct correspondance between
* the ast_cocci of julia and ast_c.
*)
let rec (split_comma: 'a wrap2 list -> ('a, il) either list) =
function
| [] -> []
| (e, ii)::xs ->
if null ii
then (Left e)::split_comma xs
else Right ii::Left e::split_comma xs
let rec (unsplit_comma: ('a, il) either list -> 'a wrap2 list) =
function
| [] -> []
| Right ii::Left e::xs ->
(e, ii)::unsplit_comma xs
| Left e::xs ->
let empty_ii = [] in
(e, empty_ii)::unsplit_comma xs
| Right ii::_ ->
raise Impossible
let split_register_param = fun (hasreg, idb, ii_b_s) ->
match hasreg, idb, ii_b_s with
| false, Some s, [i1] -> Left (s, [], i1)
| true, Some s, [i1;i2] -> Left (s, [i1], i2)
| _, None, ii -> Right ii
| _ -> raise Impossible

48
lang_cpp/parsing/todo_mly Normal file
View file

@ -0,0 +1,48 @@
%token
/*(* TTilde2? *)*/
/*(* Tunsigned Tsigned Tvoid *)*/
statement:
/*(* c++ext: TODO put at good place later *)*/
| Tswitch TOPar decl_spec init_declarator_list TCPar statement
{ StmtTodo, noii }
| Tif TOPar decl_spec init_declarator_list TCPar statement %prec LOW_PRIORITY_RULE
{ StmtTodo, noii }
| Tif TOPar decl_spec init_declarator_list TCPar statement Telse statement
{ StmtTodo, noii }
/*(* c++ext: for(int i = 0; i < n; i++)*)*/
| Tfor TOPar simple_declaration expr_statement expr_opt TCPar statement
{ StmtTodo, noii }
argument:
/* TODO: reenable, put in comment while trying to parse plan9
| action_higherordermacro { Right (ArgAction $1) }
action_higherordermacro:
| taction_list
{ if null $1
then ActMisc [Ast_cpp.fakeInfo()]
else ActMisc $1
}
*/
/* toreput, was especially used for the Linux kernel
taction_list:
(* c++ext: to remove some conflicts (from 13 to 4)
* | (* empty *) { [] }
*)
| TAny_Action { [$1] }
| taction_list TAny_Action { $1 @ [$2] }
*/

View file

@ -0,0 +1,73 @@
(* Yoann Padioleau
*
* Copyright (C) 2009 University of Urbana Champaign
*
* This program is free software; you can redistribute it and/or
* modify it under the terms of the GNU General Public License (GPL)
* version 2 as published by the Free Software Foundation.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* file license.txt for more details.
*)
(*****************************************************************************)
(* Prelude *)
(*****************************************************************************)
(*
* This file may seem redundant with the tokens generated by Yacc
* from parser.mly in parser_c.mli. The problem is that we need for
* many reasons to remember in the AST the tokens involved in the
* AST, not just the string, especially for the comment and cpp_passed
* tokens which are not in the AST at all. So,
* to avoid recursive mutual dependencies, we provide this file
* so that Ast_cpp does not need to depend on yacc which depends on
* Ast_cpp, etc.
*
* Also, ocamlyacc imposes some stupid constraints on the way we can define
* the token type. ocamlyacc forces us to do a token type that
* cant be a pair of a sum type, it must be directly a sum type.
* We don't have this constraint here.
*
* Also, some yacc tokens are not used in the grammar because they are filtered
* in some intermediate phases. But they still must be declared because
* ocamllex may generate them, or some intermediate phase may also
* generate them (like some functions in parsing_hacks.ml).
* Here we don't have this problem again so we can have a clearer token type.
*
*)
(*****************************************************************************)
(* constructs put in comments in lexer or parsing_hack *)
(*****************************************************************************)
(*
* history: was in ast_cpp.ml before:
* "This type is not in the Ast but is associated with the TCommentCpp
* token. I put this enum here because parser_c.mly needs it. I could
* have put it also in lexer_parser."
*
* update: now in token_cpp.ml, and actually right now we want those tokens
* to be in the AST so that in the matching/transforming of C code, we
* can detect if some metavariables match code which have some
* cpp_passed tokens next to them (and so where we should issue a warning).
*)
type cppcommentkind =
| CppDirective
| CppAttr
| CppMacro
| CppMacroExpanded
| CppPassingNormal (* ifdef 0, cplusplus, etc *)
| CppPassingCosWouldGetError (* expr passsing *)
(* TODO | CppPassingExplicit (* skip_start/end tag *) instead of CppOther? *)
| CppOther
(* at some point we are supposed to also parse those constructs *)
type cpluspluscommentkind =
| CplusplusTemplate
| CplusplusQualifier
(*****************************************************************************)
(* Types *)
(*****************************************************************************)

View file

@ -0,0 +1,13 @@
type cppcommentkind =
| CppDirective
| CppAttr
| CppMacro
| CppMacroExpanded
| CppPassingNormal (* ifdef 0, cplusplus, etc *)
| CppPassingCosWouldGetError (* expr passsing *)
| CppOther
type cpluspluscommentkind =
| CplusplusTemplate
| CplusplusQualifier

View file

@ -0,0 +1,730 @@
(* tokens *)
open Parser_cpp
module PI = Parse_info
(*****************************************************************************)
(* Is_xxx, categories *)
(*****************************************************************************)
let is_eof = function
| EOF _ -> true
| _ -> false
(* ---------------------------------------------------------------------- *)
let is_space = function
| TCommentSpace _ | TCommentNewline _ -> true
| _ -> false
let is_comment_or_space = function
| TCommentSpace _ | TCommentNewline _
| TComment _
-> true
| _ -> false
let is_just_comment = function
| TComment _ -> true
| _ -> false
let is_comment = function
| TCommentSpace _ | TCommentNewline _
| TComment _
| TComment_Pp _ | TComment_Cpp _
-> true
| _ -> false
let is_real_comment = function
| TComment _ | TCommentSpace _
| TCommentNewline _
-> true
| _ -> false
let is_fake_comment = function
| TComment_Pp _ | TComment_Cpp _ -> true
| _ -> false
let is_not_comment x =
not (is_comment x)
(* ---------------------------------------------------------------------- *)
(*
let is_gcc_token = function
| Tasm _ | Tinline _ | Tattribute _ | Ttypeof _
-> true
| _ -> false
*)
let is_pp_instruction = function
| TInclude _
| TDefine _
| TIfdef _ | TIfdefelse _ | TIfdefelif _
| TEndif _
| TIfdefBool _ | TIfdefMisc _ | TIfdefVersion _
| TUndef _
| TCppDirectiveOther _
-> true
| _ -> false
let is_opar = function
| TOPar _ | TOPar_Define _ | TOPar_CplusplusInit _ -> true
| _ -> false
let is_cpar = function
| TCPar _ | TCPar_EOL _ -> true
| _ -> false
let is_obrace = function
| TOBrace _ | TOBrace_DefineInit _ -> true
| _ -> false
let is_cbrace = function
| TCBrace _ -> true
| _ -> false
let is_statement = function
| Tfor _ | Tdo _ | Tif _ | Twhile _ | Treturn _
| Tbreak _ | Telse _ | Tswitch _ | Tcase _ | Tcontinue _
| Tgoto _
| TPtVirg _
| TIdent_MacroIterator _
-> true
| _ -> false
(* is_start_of_something is used in parse_c for error recovery, to find
* a synchronisation token.
*
* Would like to put TIdent or TDefine, TIfdef but they can be in the
* middle of a function, for instance with label:.
*
* Could put Typedefident but fired ? it would work in error recovery
* on the already_passed tokens, which has been already gone in the
* Parsing_hacks.lookahead machinery, but it will not work on the
* "next" tokens. But because the namespace for labels is different
* from namespace for ident/typedef, we can use the name for a typedef
* for a label and so dangerous to put Typedefident at true here.
*
* Can look in parser_c.output to know what can be at toplevel
* at the very beginning.
*)
let is_start_of_something = function
| Tchar _ | Tshort _ | Tint _ | Tdouble _ | Tfloat _ | Tlong _
| Tunsigned _ | Tsigned _ | Tvoid _
| Tauto _ | Tregister _ | Textern _ | Tstatic _
| Tconst _ | Tvolatile _
| Ttypedef _
| Tstruct _ | Tunion _ | Tenum _
(* c++ext: *)
| Tclass _
| Tbool _
| Twchar_t _
-> true
| _ -> false
let is_binary_operator = function
| TOrLog _ | TAndLog _ | TOr _ | TXor _ | TAnd _
| TEqEq _ | TNotEq _ | TInf _ | TSup _ | TInfEq _ | TSupEq _
| TShl _ | TShr _
| TPlus _ | TMinus _ | TMul _ | TDiv _ | TMod _
-> true
| _ -> false
let is_binary_operator_except_star = function
(* | TAnd _ *) (*| TMul _*)
| TOrLog _ | TAndLog _ | TOr _ | TXor _
| TEqEq _ | TNotEq _ | TInf _ | TSup _ | TInfEq _ | TSupEq _
| TShl _ | TShr _
| TPlus _ | TMinus _ | TDiv _ | TMod _
-> true
| _ -> false
let is_stuff_taking_parenthized = function
| Tif _
| Twhile _
| Tswitch _
| Ttypeof _
| TIdent_MacroIterator _
-> true
| _ -> false
let is_static_cast_like = function
| Tconst_cast _ | Tdynamic_cast _ | Tstatic_cast _ | Treinterpret_cast _ ->
true
| _ -> false
let is_basic_type = function
| Tchar _ | Tshort _ | Tint _ | Tdouble _ | Tfloat _ | Tlong _
| Tbool _ | Twchar_t _
| Tunsigned _ | Tsigned _
| Tvoid _
-> true
| _ -> false
let is_struct_like_keyword = function
| (Tstruct _ | Tunion _ | Tenum _) -> true
(* c++ext: *)
| (Tclass _) -> true
| _ -> false
let is_classkey_keyword = function
| (Tstruct _ | Tunion _ | Tclass _) -> true
| _ -> false
let is_cpp_keyword = function
| Tclass _ | Tthis _
| Tnew _
| Tdelete _
| Ttemplate _ | Ttypeid _ | Ttypename _
| Tcatch _ | Ttry _ | Tthrow _
| Toperator _
| Tpublic _ | Tprivate _ | Tprotected _
| Tfriend _
| Tvirtual _
| Tnamespace _ | Tusing _
| Tbool _
| Tfalse _ | Ttrue _
| Twchar_t _
| Tconst_cast _ | Tdynamic_cast _ | Tstatic_cast _ | Treinterpret_cast _
| Texplicit _
| Tmutable _
| Texport _
-> true
| _ -> false
let is_really_cpp_keyword = function
| Tconst_cast _ | Tdynamic_cast _ | Tstatic_cast _ | Treinterpret_cast _
-> true
(* when have some asm volatile, can have some ::
| TColCol _
-> true
*)
| _ -> false
(* some false positive on some C file like sqlite3.c *)
let is_maybenot_cpp_keyword = function
| Tpublic _ | Tprivate _ | Tprotected _
| Ttemplate _ | Tnew _ | Ttypename _
| Tnamespace _
-> true
| _ -> false
(* used in the algorithm for "10 most problematic tokens". C-s for TIdent
* in parser_cpp.mly
*)
let is_ident_like = function
| TIdent _
| TIdent_Typedef _
| TIdent_Define _
(* | TDefParamVariadic _*)
| TUnknown _
| TIdent_MacroStmt _
| TIdent_MacroString _
| TIdent_MacroIterator _
| TIdent_MacroDecl _
(* | TIdent_MacroDeclConst _ *)
(*
| TIdent_MacroAttr _
| TIdent_MacroAttrStorage _
*)
| TIdent_ClassnameInQualifier _
| TIdent_ClassnameInQualifier_BeforeTypedef _
| TIdent_Templatename _
| TIdent_TemplatenameInQualifier _
| TIdent_TemplatenameInQualifier_BeforeTypedef _
| TIdent_Constructor _
| TIdent_TypedefConstr _
-> true
| _ -> false
let is_privacy_keyword = function
| Tpublic _ | Tprivate _ | Tprotected _
-> true
| _ -> false
let token_kind_of_tok t =
match t with
(* todo: ( ) { } ... *)
| TComment _ | TComment_Pp _ | TComment_Cpp _ -> PI.Esthet PI.Comment
| TCommentSpace _ -> PI.Esthet PI.Space
| TCommentNewline _ -> PI.Esthet PI.Newline
| _ -> PI.Other
(*****************************************************************************)
(* Visitors *)
(*****************************************************************************)
(* Because ocamlyacc force us to do it that way. The ocamlyacc token
* cant be a pair of a sum type, it must be directly a sum type.
*)
let info_of_tok = function
| TString ((_s, _isWchar), i) -> i
| TChar ((_s, _isWchar), i) -> i
| TFloat ((_s, _floatType), i) -> i
| TAssign (_assignOp, i) -> i
| TIdent (_s, i) -> i
| TIdent_Typedef (_s, i) -> i
| TInt (_s, i) -> i
(*cppext:*)
| TDefine (ii) -> ii
| TInclude (_includes, _filename, i1) -> i1
| TUndef (_s, ii) -> ii
| TCppDirectiveOther (ii) -> ii
| TCommentNewline_DefineEndOfMacro (i1) -> i1
| TOPar_Define (i1) -> i1
| TIdent_Define (_s, i) -> i
| TOBrace_DefineInit (i1) -> i1
| TCppEscapedNewline (ii) -> ii
| TDefParamVariadic (_s, i1) -> i1
| TUnknown (i) -> i
| TIdent_MacroStmt (i) -> i
| TIdent_MacroString (i) -> i
| TIdent_MacroIterator (_s,i) -> i
| TIdent_MacroDecl (_s, i) -> i
| Tconst_MacroDeclConst (i) -> i
(* | TMacroTop (_s,i) -> i *)
| TCPar_EOL (i1) -> i1
| TAny_Action (i) -> i
| TComment (i) -> i
| TCommentSpace (i) -> i
| TComment_Pp (_cppkind, i) -> i
| TComment_Cpp (_cppkind, i) -> i
| TCommentNewline (i) -> i
| TIfdef (i) -> i
| TIfdefelse (i) -> i
| TIfdefelif (i) -> i
| TEndif (i) -> i
| TIfdefBool (_b, i) -> i
| TIfdefMisc (_b, i) -> i
| TIfdefVersion (_b, i) -> i
| TOPar (i) -> i
| TOPar_CplusplusInit (i) -> i
| TCPar (i) -> i
| TOBrace (i) -> i
| TCBrace (i) -> i
| TOCro (i) -> i
| TCCro (i) -> i
| TDot (i) -> i
| TComma (i) -> i
| TPtrOp (i) -> i
| TInc (i) -> i
| TDec (i) -> i
| TEq (i) -> i
| TWhy (i) -> i
| TTilde (i) -> i
| TBang (i) -> i
| TEllipsis (i) -> i
| TCol (i) -> i
| TPtVirg (i) -> i
| TOrLog (i) -> i
| TAndLog (i) -> i
| TOr (i) -> i
| TXor (i) -> i
| TAnd (i) -> i
| TEqEq (i) -> i
| TNotEq (i) -> i
| TInf (i) -> i
| TSup (i) -> i
| TInfEq (i) -> i
| TSupEq (i) -> i
| TShl (i) -> i
| TShr (i) -> i
| TPlus (i) -> i
| TMinus (i) -> i
| TMul (i) -> i
| TDiv (i) -> i
| TMod (i) -> i
| Tchar (i) -> i
| Tshort (i) -> i
| Tint (i) -> i
| Tdouble (i) -> i
| Tfloat (i) -> i
| Tlong (i) -> i
| Tunsigned (i) -> i
| Tsigned (i) -> i
| Tvoid (i) -> i
| Tauto (i) -> i
| Tregister (i) -> i
| Textern (i) -> i
| Tstatic (i) -> i
| Tconst (i) -> i
| Tvolatile (i) -> i
| Trestrict (i) -> i
| Tstruct (i) -> i
| Tenum (i) -> i
| Ttypedef (i) -> i
| Tunion (i) -> i
| Tbreak (i) -> i
| Telse (i) -> i
| Tswitch (i) -> i
| Tcase (i) -> i
| Tcontinue (i) -> i
| Tfor (i) -> i
| Tdo (i) -> i
| Tif (i) -> i
| Twhile (i) -> i
| Treturn (i) -> i
| Tgoto (i) -> i
| Tdefault (i) -> i
| Tsizeof (i) -> i
(* gccext: *)
| Tasm (i) -> i
| Tattribute (i) -> i
| Tinline (i) -> i
| Ttypeof (i) -> i
(* c++ext: *)
| Tclass (i) -> i
| Tthis (i) -> i
| Tnew (i) -> i
| Tdelete (i) -> i
| Ttemplate (i) -> i
| Ttypeid (i) -> i
| Ttypename (i) -> i
| Tcatch (i) -> i
| Ttry (i) -> i
| Tthrow (i) -> i
| Toperator (i) -> i
| Tpublic (i) -> i
| Tprivate (i) -> i
| Tprotected (i) -> i
| Tfriend (i) -> i
| Tvirtual (i) -> i
| Tnamespace (i) -> i
| Tusing (i) -> i
| Tbool (i) -> i
| Ttrue (i) -> i
| Tfalse (i) -> i
| Twchar_t (i) -> i
| Tconst_cast (i) -> i
| Tdynamic_cast (i) -> i
| Tstatic_cast (i) -> i
| Treinterpret_cast (i) -> i
| Texplicit (i) -> i
| Tmutable (i) -> i
| Texport (i) -> i
| TColCol (i) -> i
| TColCol_BeforeTypedef (i) -> i
| TPtrOpStar (i) -> i
| TDotStar(i) -> i
| TIdent_ClassnameInQualifier (_s, i) -> i
| TIdent_ClassnameInQualifier_BeforeTypedef (_s, i) -> i
| TIdent_Templatename (_s, i) -> i
| TIdent_Constructor (_s, i) -> i
| TIdent_TypedefConstr (_s, i) -> i
| TIdent_TemplatenameInQualifier (_s, i) -> i
| TIdent_TemplatenameInQualifier_BeforeTypedef (_s, i) -> i
| TInf_Template (i) -> i
| TSup_Template (i) -> i
| TOCro_new (i) -> i
| TCCro_new (i) -> i
| TInt_ZeroVirtual (i) -> i
| Tchar_Constr (i) -> i
| Tint_Constr (i) -> i
| Tfloat_Constr (i) -> i
| Tdouble_Constr (i) -> i
| Twchar_t_Constr (i) -> i
| Tshort_Constr (i) -> i
| Tlong_Constr (i) -> i
| Tbool_Constr (i) -> i
| Tunsigned_Constr i -> i
| Tsigned_Constr i -> i
| EOF (i) -> i
(* used by tokens to complete the parse_info with filename, line, col infos *)
let visitor_info_of_tok f = function
| TString ((s, isWchar), i) -> TString ((s, isWchar), f i)
| TChar ((s, isWchar), i) -> TChar ((s, isWchar), f i)
| TFloat ((s, floatType), i) -> TFloat ((s, floatType), f i)
| TAssign (assignOp, i) -> TAssign (assignOp, f i)
| TIdent (s, i) -> TIdent (s, f i)
| TIdent_Typedef (s, i) -> TIdent_Typedef (s, f i)
| TInt (s, i) -> TInt (s, f i)
(* cppext: *)
| TDefine (i1) -> TDefine(f i1)
| TUndef (s,i1) -> TUndef(s, f i1)
| TCppDirectiveOther (i1) -> TCppDirectiveOther(f i1)
| TInclude (includes, filename, i1) ->
TInclude (includes, filename, f i1)
| TCppEscapedNewline (i1) -> TCppEscapedNewline (f i1)
| TCommentNewline_DefineEndOfMacro (i1) ->
TCommentNewline_DefineEndOfMacro (f i1)
| TOPar_Define (i1) -> TOPar_Define (f i1)
| TIdent_Define (s, i) -> TIdent_Define (s, f i)
| TDefParamVariadic (s, i1) -> TDefParamVariadic (s, f i1)
| TOBrace_DefineInit (i1) -> TOBrace_DefineInit (f i1)
| TUnknown (i) -> TUnknown (f i)
| TIdent_MacroStmt (i) -> TIdent_MacroStmt (f i)
| TIdent_MacroString (i) -> TIdent_MacroString (f i)
| TIdent_MacroIterator (s,i) -> TIdent_MacroIterator (s,f i)
| TIdent_MacroDecl (s,i) -> TIdent_MacroDecl (s, f i)
| Tconst_MacroDeclConst (i) -> Tconst_MacroDeclConst (f i)
(* | TMacroTop (s,i) -> TMacroTop (s,f i) *)
| TCPar_EOL (i) -> TCPar_EOL (f i)
| TAny_Action (i) -> TAny_Action (f i)
| TComment (i) -> TComment (f i)
| TCommentSpace (i) -> TCommentSpace (f i)
| TCommentNewline (i) -> TCommentNewline (f i)
| TComment_Pp (cppkind, i) -> TComment_Pp (cppkind, f i)
| TComment_Cpp (cppkind, i) -> TComment_Cpp (cppkind, f i)
| TIfdef (i) -> TIfdef (f i)
| TIfdefelse (i) -> TIfdefelse (f i)
| TIfdefelif (i) -> TIfdefelif (f i)
| TEndif (i) -> TEndif (f i)
| TIfdefBool (b, i) -> TIfdefBool (b, f i)
| TIfdefMisc (b, i) -> TIfdefMisc (b, f i)
| TIfdefVersion (b, i) -> TIfdefVersion (b, f i)
| TOPar (i) -> TOPar (f i)
| TOPar_CplusplusInit (i) -> TOPar_CplusplusInit (f i)
| TCPar (i) -> TCPar (f i)
| TOBrace (i) -> TOBrace (f i)
| TCBrace (i) -> TCBrace (f i)
| TOCro (i) -> TOCro (f i)
| TCCro (i) -> TCCro (f i)
| TDot (i) -> TDot (f i)
| TComma (i) -> TComma (f i)
| TPtrOp (i) -> TPtrOp (f i)
| TInc (i) -> TInc (f i)
| TDec (i) -> TDec (f i)
| TEq (i) -> TEq (f i)
| TWhy (i) -> TWhy (f i)
| TTilde (i) -> TTilde (f i)
| TBang (i) -> TBang (f i)
| TEllipsis (i) -> TEllipsis (f i)
| TCol (i) -> TCol (f i)
| TPtVirg (i) -> TPtVirg (f i)
| TOrLog (i) -> TOrLog (f i)
| TAndLog (i) -> TAndLog (f i)
| TOr (i) -> TOr (f i)
| TXor (i) -> TXor (f i)
| TAnd (i) -> TAnd (f i)
| TEqEq (i) -> TEqEq (f i)
| TNotEq (i) -> TNotEq (f i)
| TInf (i) -> TInf (f i)
| TSup (i) -> TSup (f i)
| TInfEq (i) -> TInfEq (f i)
| TSupEq (i) -> TSupEq (f i)
| TShl (i) -> TShl (f i)
| TShr (i) -> TShr (f i)
| TPlus (i) -> TPlus (f i)
| TMinus (i) -> TMinus (f i)
| TMul (i) -> TMul (f i)
| TDiv (i) -> TDiv (f i)
| TMod (i) -> TMod (f i)
| Tchar (i) -> Tchar (f i)
| Tshort (i) -> Tshort (f i)
| Tint (i) -> Tint (f i)
| Tdouble (i) -> Tdouble (f i)
| Tfloat (i) -> Tfloat (f i)
| Tlong (i) -> Tlong (f i)
| Tunsigned (i) -> Tunsigned (f i)
| Tsigned (i) -> Tsigned (f i)
| Tvoid (i) -> Tvoid (f i)
| Tauto (i) -> Tauto (f i)
| Tregister (i) -> Tregister (f i)
| Textern (i) -> Textern (f i)
| Tstatic (i) -> Tstatic (f i)
| Tconst (i) -> Tconst (f i)
| Tvolatile (i) -> Tvolatile (f i)
| Trestrict (i) -> Trestrict (f i)
| Tstruct (i) -> Tstruct (f i)
| Tenum (i) -> Tenum (f i)
| Ttypedef (i) -> Ttypedef (f i)
| Tunion (i) -> Tunion (f i)
| Tbreak (i) -> Tbreak (f i)
| Telse (i) -> Telse (f i)
| Tswitch (i) -> Tswitch (f i)
| Tcase (i) -> Tcase (f i)
| Tcontinue (i) -> Tcontinue (f i)
| Tfor (i) -> Tfor (f i)
| Tdo (i) -> Tdo (f i)
| Tif (i) -> Tif (f i)
| Twhile (i) -> Twhile (f i)
| Treturn (i) -> Treturn (f i)
| Tgoto (i) -> Tgoto (f i)
| Tdefault (i) -> Tdefault (f i)
| Tsizeof (i) -> Tsizeof (f i)
| Tasm (i) -> Tasm (f i)
| Tattribute (i) -> Tattribute (f i)
| Tinline (i) -> Tinline (f i)
| Ttypeof (i) -> Ttypeof (f i)
| Tclass (i) -> Tclass (f i)
| Tthis (i) -> Tthis (f i)
| Tnew (i) -> Tnew (f i)
| Tdelete (i) -> Tdelete (f i)
| Ttemplate (i) -> Ttemplate (f i)
| Ttypeid (i) -> Ttypeid (f i)
| Ttypename (i) -> Ttypename (f i)
| Tcatch (i) -> Tcatch (f i)
| Ttry (i) -> Ttry (f i)
| Tthrow (i) -> Tthrow (f i)
| Toperator (i) -> Toperator (f i)
| Tpublic (i) -> Tpublic (f i)
| Tprivate (i) -> Tprivate (f i)
| Tprotected (i) -> Tprotected (f i)
| Tfriend (i) -> Tfriend (f i)
| Tvirtual (i) -> Tvirtual (f i)
| Tnamespace (i) -> Tnamespace (f i)
| Tusing (i) -> Tusing (f i)
| Tbool (i) -> Tbool (f i)
| Ttrue (i) -> Ttrue (f i)
| Tfalse (i) -> Tfalse (f i)
| Twchar_t (i) -> Twchar_t (f i)
| Tconst_cast (i) -> Tconst_cast (f i)
| Tdynamic_cast (i) -> Tdynamic_cast (f i)
| Tstatic_cast (i) -> Tstatic_cast (f i)
| Treinterpret_cast (i) -> Treinterpret_cast (f i)
| Texplicit (i) -> Texplicit (f i)
| Tmutable (i) -> Tmutable (f i)
| Texport (i) -> Texport (f i)
| TColCol (i) -> TColCol (f i)
| TColCol_BeforeTypedef (i) -> TColCol_BeforeTypedef (f i)
| TPtrOpStar (i) -> TPtrOpStar (f i)
| TDotStar(i) -> TDotStar (f i)
| TIdent_ClassnameInQualifier (s, i) -> TIdent_ClassnameInQualifier (s, f i)
| TIdent_ClassnameInQualifier_BeforeTypedef (s, i) ->
TIdent_ClassnameInQualifier_BeforeTypedef (s, f i)
| TIdent_Templatename (s, i) -> TIdent_Templatename (s, f i)
| TIdent_Constructor (s, i) -> TIdent_Constructor (s, f i)
| TIdent_TypedefConstr (s, i) -> TIdent_TypedefConstr (s, f i)
| TIdent_TemplatenameInQualifier (s, i) ->
TIdent_TemplatenameInQualifier (s, f i)
| TIdent_TemplatenameInQualifier_BeforeTypedef (s, i) ->
TIdent_TemplatenameInQualifier_BeforeTypedef (s, f i)
| TInf_Template (i) -> TInf_Template (f i)
| TSup_Template (i) -> TSup_Template (f i)
| TOCro_new (i) -> TOCro_new (f i)
| TCCro_new (i) -> TCCro_new (f i)
| TInt_ZeroVirtual (i) -> TInt_ZeroVirtual (f i)
| Tchar_Constr (i) -> Tchar_Constr (f i)
| Tint_Constr (i) -> Tint_Constr (f i)
| Tfloat_Constr (i) -> Tfloat_Constr (f i)
| Tdouble_Constr (i) -> Tdouble_Constr (f i)
| Twchar_t_Constr (i) -> Twchar_t_Constr (f i)
| Tshort_Constr (i) -> Tshort_Constr (f i)
| Tlong_Constr (i) -> Tlong_Constr (f i)
| Tbool_Constr (i) -> Tbool_Constr (f i)
| Tsigned_Constr (i) -> Tsigned_Constr (f i)
| Tunsigned_Constr (i) -> Tunsigned_Constr (f i)
| EOF (i) -> EOF (f i)
(*****************************************************************************)
(* Accessors *)
(*****************************************************************************)
let line_of_tok tok =
let info = info_of_tok tok in
PI.line_of_info info

View file

@ -0,0 +1,41 @@
val is_space : Parser_cpp.token -> bool
val is_comment_or_space : Parser_cpp.token -> bool
val is_just_comment : Parser_cpp.token -> bool
val is_comment : Parser_cpp.token -> bool
val is_real_comment : Parser_cpp.token -> bool
val is_fake_comment : Parser_cpp.token -> bool
val is_not_comment : Parser_cpp.token -> bool
val is_pp_instruction : Parser_cpp.token -> bool
val is_eof : Parser_cpp.token -> bool
val is_statement : Parser_cpp.token -> bool
val is_start_of_something : Parser_cpp.token -> bool
val is_binary_operator : Parser_cpp.token -> bool
val is_stuff_taking_parenthized : Parser_cpp.token -> bool
val is_static_cast_like : Parser_cpp.token -> bool
val is_basic_type : Parser_cpp.token -> bool
val is_binary_operator_except_star : Parser_cpp.token -> bool
val is_struct_like_keyword : Parser_cpp.token -> bool
val is_classkey_keyword : Parser_cpp.token -> bool
val is_cpp_keyword : Parser_cpp.token -> bool
val is_really_cpp_keyword : Parser_cpp.token -> bool
val is_maybenot_cpp_keyword : Parser_cpp.token -> bool
val is_privacy_keyword: Parser_cpp.token -> bool
val is_opar : Parser_cpp.token -> bool
val is_cpar : Parser_cpp.token -> bool
val is_obrace : Parser_cpp.token -> bool
val is_cbrace : Parser_cpp.token -> bool
val is_ident_like: Parser_cpp.token -> bool
val token_kind_of_tok: Parser_cpp.token -> Parse_info.token_kind
val info_of_tok :
Parser_cpp.token -> Parse_info.info
val visitor_info_of_tok :
(Parse_info.info -> Parse_info.info) -> Parser_cpp.token -> Parser_cpp.token
val line_of_tok : Parser_cpp.token -> int

View file

@ -0,0 +1,327 @@
(* Yoann Padioleau
*
* Copyright (C) 2014 Facebook
* Copyright (C) 2002-2008 Yoann Padioleau
*
* This program is free software; you can redistribute it and/or
* modify it under the terms of the GNU General Public License (GPL)
* version 2 as published by the Free Software Foundation.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* file license.txt for more details.
*)
open Common
open Parser_cpp
open Token_views_cpp
module TH = Token_helpers_cpp
module TV = Token_views_cpp
(*****************************************************************************)
(* Prelude *)
(*****************************************************************************)
(*****************************************************************************)
(* Argument vs Parameter *)
(*****************************************************************************)
let look_like_argument _tok_before xs =
(* normalize for C++ *)
let xs = xs +> List.map (function
| Tok ({t=TAnd ii} as record) -> Tok ({record with t=TMul ii})
| x -> x
)
in
(* split by comma so can easily check if have stuff like '*xx'
* that takes the full argument
*)
let xxs = split_comma xs in
let aux1 xs =
match xs with
| [] -> false
(* *xx (note: actually can also be a function pointer decl) *)
| [Tok{t=TMul _}; Tok{t=TIdent _}] -> true
(* *(xx) *)
| [Tok{t=TMul _}; Parens _] -> true
(* TODO: xx * yy and space = 1 between the 2 :) *)
| _ -> false
in
let rec aux xs =
match xs with
| [] -> false
(* a function call probably *)
| Tok{t=TIdent _}::Parens _::_xs ->
(* todo? look_like_argument recursively in Parens || aux xs ? *)
true
(* if have = ... then must stop, could be default parameter of a method *)
| Tok{t=TEq _}::_xs -> false
(* could be part of a type declaration *)
| Tok {t=TOCro _}::Tok {t=TCCro _}::_xs -> false
| Tok {t=TOCro _}::Tok {t=(TInt _)}::Tok {t=TCCro _}::_xs -> false
| Tok {t=TOCro _}::Tok {t=(TIdent _)}::Tok {t=TCCro _}::_xs -> false
| x::xs ->
(match x with
| Tok {t=(TInt _ | TFloat _ | TChar _ | TString _) } -> true
| Tok {t=(Ttrue _ | Tfalse _) } -> true
| Tok {t=(Tthis _)} -> true
| Tok {t=(Tnew _ )} -> true
| Tok {t= tok} when TH.is_binary_operator_except_star tok -> true
| Tok {t=(TInc _ | TDec _)} -> true
| Tok {t = (TDot _ | TPtrOp _ | TPtrOpStar _ | TDotStar _)} -> true
| Tok {t = (TOCro _)} -> true
| Tok {t = (TWhy _ | TBang _)} -> true
| _ -> aux xs
)
in
(* todo? what if they contradict each other? if one say arg and
* the other a parameter?
*)
xxs +> List.exists aux1 || aux xs
let look_like_typedef s =
s =~ ".*_t$" ||
s = "ulong" || s = "uchar" || s = "uvlong" || s = "vlong" || s = "uintptr"
(* plan9, but actually some fp such as Paddr which is actually a macro *)
(* || s =~ "[A-Z][a-z].*$" *)
(* with DECLARE_BOOST_TYPE, but have some false positives
* when people do xx* indexPtr = const_cast<>(indexPtr);
*)
(* s =~ ".*Ptr$" *)
(* || s = "StringPiece" *)
(* todo: pass1, look for const, etc
* todo: pass2, look xx_t, xx&, xx*, xx**, see heuristics in typedef
*
* Many patterns should mimic some heuristics in parsing_hack_typedef.ml
*)
let look_like_parameter tok_before xs =
(* normalize for C++ *)
let xs = xs +> List.map (function
| Tok ({t=TAnd ii} as record) -> Tok ({record with t=TMul ii})
| x -> x
)
in
let xxs = split_comma xs in
let aux1 xs =
match xs with
| [] -> false
(* xx_t *)
| [Tok {t=TIdent (s, _)}] when look_like_typedef s -> true
(* xx* *)
| [Tok {t=TIdent _}; Tok {t=TMul _}] -> true
(* xx** *)
| [Tok {t=TIdent _}; Tok {t=TMul _}; Tok {t=TMul _}] -> true
(* xx * y could be multiplication (or xx & yy) ..
* todo: could look if space around :) but because of the
* filtering of template and qualifier the no_space_between
* may not be completely accurate here. May need lower level access
* to the list of TCommentSpace and their position.
* hmm but can look at col?
*
* C-s for parameter_decl in grammar to see that catch() is
* a InParameter.
*)
| [Tok {t=TIdent _}; Tok {t=TMul _};Tok {t=TIdent _};] ->
(match tok_before with
| Tok{t=(
Tcatch _
(* ugly: TIdent_Constructor interaction between past heuristics *)
| TIdent_Constructor _
| Toperator _
(* no! | TIdent _ *)
)} -> true
| _ -> false
)
| _ -> false
in
let rec aux xs =
match xs with
| [] -> false
(* xx yy *)
| Tok {t=TIdent _}::Tok{t=TIdent _}::_xs -> true
| x::xs ->
(match x with
| Tok {t= tok} when TH.is_basic_type tok -> true
| Tok {t = (Tconst _ | Tvolatile _)} -> true
| Tok {t = (Tstruct _ | Tunion _ | Tenum _ | Tclass _)} -> true
| _ -> aux xs
)
in
xxs +> List.exists aux1 || aux xs
(*****************************************************************************)
(* Main heuristics *)
(*****************************************************************************)
(*
* Most of the important contexts are introduced via some '{' '}'. To
* disambiguate is it often enough to just look at a few tokens before the
* '{'.
*
* Below we assume a view without:
* - comments
* - cpp directives
*
* todo
* - handle more C++ (right now I did it mostly to be able to parse plan9)
* - harder now that have c++, can have function inside struct so need
* handle all together.
* - change token but do not recurse in
* nested Braceised. maybe do via accumulator, don't use iter_token_brace?
* - need remove the qualifier as they make the sequence pattern matching
* more difficult?
*)
let set_context_tag_multi groups =
let rec aux xs =
match xs with
| [] -> ()
(* struct Foo {, also valid for class and union *)
| Tok{t=(Tstruct _ | Tunion _ | Tclass _)}::Tok{t=TIdent(s,_)}
::(Braces(_t1, _body, _t2) as braces)::xs
->
[braces] +> TV.iter_token_multi (fun tok ->
tok.TV.where <- (TV.InClassStruct s)::tok.TV.where;
);
aux (braces::xs)
| Tok{t=(Tstruct _ | Tunion _)}::(Braces(_t1, _body, _t2) as braces)::xs
->
[braces] +> TV.iter_token_multi (fun tok ->
tok.TV.where <- (TV.InClassStruct "__anon__")::tok.TV.where;
);
aux (braces::xs)
(* = { } *)
| Tok ({t=TEq _; _})::(Braces(_t1, _body, _t2) as braces)::xs ->
[braces] +> TV.iter_token_multi (fun tok ->
tok.TV.where <- InInitializer::tok.TV.where;
);
aux (braces::xs)
(* enum xxx { InEnum *)
| Tok{t=Tenum _}::Tok{t=TIdent(_,_)}::(Braces(_t1, _body, _t2) as braces)::xs
| Tok{t=Tenum _}::(Braces(_t1, _body, _t2) as braces)::xs
->
[braces] +> TV.iter_token_multi (fun tok ->
tok.TV.where <- TV.InEnum::tok.TV.where;
);
aux (braces::xs)
(* C++: class Foo : ... { *)
| Tok{t=Tclass _ | Tstruct _}::Tok{t=TIdent(s,_)}
::Tok{t= TCol ii}::xs
->
let (before, braces, after) =
try
xs +> Common2.split_when (function
| Braces _ -> true
| _ -> false
)
with Not_found ->
raise (UnclosedSymbol (spf "PB with split_when at %s"
(Parse_info.string_of_info ii)))
in
aux before;
[braces] +> TV.iter_token_multi (fun tok ->
tok.TV.where <- (TV.InClassStruct s)::tok.TV.where;
);
aux [braces];
aux after
(* need to look what was before to help the look_like_xxx heuristics
*
* The order of the 3 rules below is important. We must first try
* look_like_argument which has less FP than look_like_parameter
*)
| x::(Parens(_t1, body, _t2) as parens)::xs
when look_like_argument x body ->
(*msg_context t1.t (TV.InArgument); *)
[parens] +> TV.iter_token_multi (fun tok ->
tok.TV.where <- (TV.InArgument)::tok.TV.where;
);
(* todo? recurse on body? *)
aux [x];
aux (parens::xs)
(* C++: special cases *)
| (Tok{t=Toperator _} as tok1)::tok2::(Parens(_t1, body, _t2) as parens)::xs
when look_like_parameter tok1 body ->
(* msg_context t1.t (TV.InParameter); *)
[parens] +> TV.iter_token_multi (fun tok ->
tok.TV.where <- (TV.InParameter)::tok.TV.where;
);
(* recurse on body? hmm if InParameter should not have nested
* stuff except when pass function pointer
*)
aux [tok1;tok2];
aux (parens::xs)
| x::(Parens(_t1, body, _t2) as parens)::xs
when look_like_parameter x body ->
(* msg_context t1.t (TV.InParameter); *)
[parens] +> TV.iter_token_multi (fun tok ->
tok.TV.where <- (TV.InParameter)::tok.TV.where;
);
(* recurse on body? hmm if InParameter should not have nested
* stuff except when pass function pointer
*)
aux [x];
aux (parens::xs)
(* void xx() *)
| Tok{t=typ}::Tok{t=TIdent _}::(Parens(_t1, _body, _t2) as parens)::xs
when TH.is_basic_type typ ->
(* msg_context t1.t (TV.InParameter); *)
[parens] +> TV.iter_token_multi (fun tok ->
tok.TV.where <- (TV.InParameter)::tok.TV.where;
);
aux (parens::xs)
| x::xs ->
(match x with
| Tok _t -> ()
| Parens (_t1, xs, _t2)
| Braces (_t1, xs, _t2)
| Angle (_t1, xs, _t2)
->
aux xs
);
aux xs
in
(* sane initialization *)
groups +> TV.iter_token_multi (fun tok ->
tok.TV.where <- [TV.InTopLevel];
);
aux groups
(*****************************************************************************)
(* Main heuristics C++ *)
(*****************************************************************************)
(*
* assumes a view without:
* - template arguments, qualifiers,
* - comments and cpp directives
* - TODO public/protected/... ?
*)
let set_context_tag_cplus groups =
set_context_tag_multi groups

View file

@ -0,0 +1,10 @@
val set_context_tag_cplus:
Token_views_cpp.multi_grouped list -> unit
val set_context_tag_multi:
Token_views_cpp.multi_grouped list -> unit
(* todo: could be moved *)
val look_like_typedef:
string -> bool

View file

@ -0,0 +1,665 @@
(* Yoann Padioleau
*
* Copyright (C) 2011, 2014 Facebook
* Copyright (C) 2007, 2008 Ecole des Mines de Nantes
*
* This program is free software; you can redistribute it and/or
* modify it under the terms of the GNU General Public License (GPL)
* version 2 as published by the Free Software Foundation.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* file license.txt for more details.
*)
open Common
module Flag = Flag_parsing_cpp
module PI = Parse_info
module TH = Token_helpers_cpp
open Parser_cpp
(*****************************************************************************)
(* Prelude *)
(*****************************************************************************)
(*
* This module makes it easier to write some fuzzy parsing heuristics
* by offering different "views" over the same set of tokens.
*
* Normally I should not use ref/mutable in the token_extended type below
* and instead have a set of functions taking a list of tokens and
* returning a list of tokens. The problem is that to make easier some
* functions, it is better to work on better representation, on "views"
* over this list of tokens. But then modifying those views and get
* back from those views to the original simple list of tokens is
* tedious. One way is to maintain next to the view a list of "actions"
* (I was using a hash storing the charpos of the token and associating
* the action) but it is tedious too. Simpler to use mutable/ref. We
* use the same idea that we use when working on the Ast.
*
* old: when I was using the list of "actions" next to the views, the hash
* indexed by the charpos, there could have been some problems:
* how my fake_pos interact with the way I tag and adjust token ?
* because I base my tagging on the position of the token ! so sometimes
* could tag another fakeInfo that should not be tagged ?
* fortunately I don't use anymore this technique.
*)
(*****************************************************************************)
(* Some debugging functions *)
(*****************************************************************************)
let pr2, _pr2_once = Common2.mk_pr2_wrappers Flag.verbose_parsing
(*****************************************************************************)
(* Types *)
(*****************************************************************************)
type token_extended = {
(* chose 't' and not 'tok' to have a short name because we will write
* lots of ocaml patterns around this ... so better to be short
*)
mutable t: Parser_cpp.token;
(* In C++ we have functions inside classes, so need a stack of context *)
mutable where: context list;
(* less: need also a after ? *)
mutable new_tokens_before : Parser_cpp.token list;
(* line x col cache (more easily accessible) of the info in the token *)
line: int;
col : int;
}
(* The strategy to tag is mostly to look at the token(s) before the '{' *)
and context =
| InTopLevel
| InClassStruct of string (* can be __anon__ *) | InEnum
| InInitializer
| InAssign
| InParameter | InArgument
(* TODO actually commented in token_view_context because of c++ *)
| InFunction
(*
| InTemplateParam (* TODO *)
*)
(* InCondition ? InParenExpr ? *)
(* x list list, because x list separated by ',' *)
type paren_grouped =
| Parenthised of paren_grouped list list * token_extended list
| PToken of token_extended
type brace_grouped =
| Braceised of
brace_grouped list list * token_extended * token_extended option
| BToken of token_extended
(* Far better data structure than doing hacks in the lexer or parser
* because in lexer we don't know to which ifdef a endif is related
* and so when we want to comment a ifdef, we don't know which endif
* we must also comment. Especially true for the #if 0 which sometimes
* have a #else part.
*
* x list list, because x list separated by #else or #elif
*)
type ifdef_grouped =
| Ifdef of ifdef_grouped list list * token_extended list
| Ifdefbool of bool * ifdef_grouped list list * token_extended list
| NotIfdefLine of token_extended list
type 'a line_grouped =
Line of 'a list
type body_function_grouped =
| BodyFunction of token_extended list
| NotBodyLine of token_extended list
(* quite similar to ast_fuzzy.ml but with extended token *)
type multi_grouped =
| Braces of token_extended * multi_grouped list * token_extended option
| Parens of token_extended * multi_grouped list * token_extended option
| Angle of token_extended * multi_grouped list * token_extended option
| Tok of token_extended
(* with tarzan *)
exception UnclosedSymbol of string
(*****************************************************************************)
(* Helpers *)
(*****************************************************************************)
let mk_token_extended x =
let info = TH.info_of_tok x in
let (line, col) = PI.line_of_info info, PI.col_of_info info in
{ t = x;
line = line; col = col;
(* we use List.hd at a few places, so convenient to have a sentinel *)
where = [InTopLevel];
new_tokens_before = [];
}
let mk_token_fake x =
{ t = x;
line = -1; col = -1;
where = [InTopLevel];
new_tokens_before = [];
}
let rebuild_tokens_extented toks_ext =
let _tokens = ref [] in
toks_ext +> List.iter (fun tok ->
tok.new_tokens_before +> List.iter (fun x -> push x _tokens);
push tok.t _tokens
);
let tokens = List.rev !_tokens in
(tokens +> Common2.acc_map mk_token_extended)
(*****************************************************************************)
(* View builders *)
(*****************************************************************************)
(* ------------------------------------------------------------------------- *)
(* Parens *)
(* ------------------------------------------------------------------------- *)
(* todo: synchro ! use more indentation
* if paren not closed and same indentation level, certainly because
* part of a mid-ifdef-expression.
*
* c++ext: TODO: need to handle templates here.
* The parenthized view must not consider the ',' in expressions
* like foo(lexical cast<string,int>, ...) as a separator for the arguments
* of foo(), otherwise we will get [lexical_cast<string; ...] which
* could confuse some heuristics.
*
* pre: have done the TInf->TInf_Template translation.
*)
let rec mk_parenthised xs =
match xs with
| [] -> []
| x::xs ->
(match x.t with
| xx when TH.is_opar xx ->
let body, extras, xs = mk_parameters [x] [] xs in
Parenthised (body,extras)::mk_parenthised xs
| _ ->
PToken x::mk_parenthised xs
)
(* return the body of the parenthised expression and the rest of the tokens *)
and mk_parameters extras acc_before_sep xs =
match xs with
| [] ->
(* maybe because of #ifdef which "opens" '(' in 2 branches *)
pr2 "PB: not found closing paren in fuzzy parsing";
[List.rev acc_before_sep], List.rev extras, []
| x::xs ->
(match x.t with
(* synchro *)
| xx when TH.is_obrace xx && x.col = 0 ->
pr2 "PB: found synchro point } in paren";
[List.rev acc_before_sep], List.rev (extras), (x::xs)
| xx when TH.is_cpar xx ->
[List.rev acc_before_sep], List.rev (x::extras), xs
| xx when TH.is_opar xx ->
let body, extrasnest, xs = mk_parameters [x] [] xs in
mk_parameters extras
(Parenthised (body,extrasnest)::acc_before_sep)
xs
| TComma _ ->
let body, extras, xs = mk_parameters (x::extras) [] xs in
(List.rev acc_before_sep)::body, extras, xs
| _ ->
mk_parameters extras (PToken x::acc_before_sep) xs
)
(* ------------------------------------------------------------------------- *)
(* Brace *)
(* ------------------------------------------------------------------------- *)
let rec mk_braceised xs =
match xs with
| [] -> []
| x::xs ->
(match x.t with
| xx when TH.is_obrace xx ->
let body, endbrace, xs = mk_braceised_aux [] xs in
Braceised (body, x, endbrace)::mk_braceised xs
| xx when TH.is_cbrace xx ->
pr2 "PB: found closing brace alone in fuzzy parsing";
BToken x::mk_braceised xs
| _ ->
BToken x::mk_braceised xs
)
(* return the body of the parenthised expression and the rest of the tokens *)
and mk_braceised_aux acc xs =
match xs with
| [] ->
(* maybe because of #ifdef which "opens" '(' in 2 branches *)
pr2 "PB: not found closing brace in fuzzy parsing";
[List.rev acc], None, []
| x::xs ->
(match x.t with
| xx when TH.is_cbrace xx -> [List.rev acc], Some x, xs
| xx when TH.is_obrace xx ->
let body, endbrace, xs = mk_braceised_aux [] xs in
mk_braceised_aux (Braceised (body,x, endbrace)::acc) xs
| _ ->
mk_braceised_aux (BToken x::acc) xs
)
(* ------------------------------------------------------------------------- *)
(* Ifdefs *)
(* ------------------------------------------------------------------------- *)
let rec mk_ifdef xs =
match xs with
| [] -> []
| x::xs ->
(match x.t with
| TIfdef _ ->
let body, extra, xs = mk_ifdef_parameters [x] [] xs in
Ifdef (body, extra)::mk_ifdef xs
| TIfdefBool (b,_) ->
let body, extra, xs = mk_ifdef_parameters [x] [] xs in
(* if not passing, then consider a #if 0 as an ordinary #ifdef *)
if !Flag.if0_passing
then Ifdefbool (b, body, extra)::mk_ifdef xs
else Ifdef(body, extra)::mk_ifdef xs
| TIfdefMisc (b,_) | TIfdefVersion (b,_) ->
let body, extra, xs = mk_ifdef_parameters [x] [] xs in
Ifdefbool (b, body, extra)::mk_ifdef xs
| _ ->
(* todo? can have some Ifdef in the line ? *)
let line, xs = Common.span (fun y -> y.line = x.line) (x::xs) in
NotIfdefLine line::mk_ifdef xs
)
and mk_ifdef_parameters extras acc_before_sep xs =
match xs with
| [] ->
(* Note that mk_ifdef is assuming that CPP instruction are alone
* on their line. Because I do a span (fun x -> is_same_line ...)
* I might take with me a #endif if this one is mixed on a line
* with some "normal" tokens.
*)
pr2 "PB: not found closing ifdef in fuzzy parsing";
[List.rev acc_before_sep], List.rev extras, []
| x::xs ->
(match x.t with
| TEndif _ ->
[List.rev acc_before_sep], List.rev (x::extras), xs
| TIfdef _ ->
let body, extrasnest, xs = mk_ifdef_parameters [x] [] xs in
mk_ifdef_parameters
extras (Ifdef (body, extrasnest)::acc_before_sep) xs
| TIfdefBool (b,_) ->
let body, extrasnest, xs = mk_ifdef_parameters [x] [] xs in
if !Flag.if0_passing
then
mk_ifdef_parameters
extras (Ifdefbool (b, body, extrasnest)::acc_before_sep) xs
else
mk_ifdef_parameters
extras (Ifdef (body, extrasnest)::acc_before_sep) xs
| TIfdefMisc (b,_) | TIfdefVersion (b,_) ->
let body, extrasnest, xs = mk_ifdef_parameters [x] [] xs in
mk_ifdef_parameters
extras (Ifdefbool (b, body, extrasnest)::acc_before_sep) xs
| TIfdefelse _
| TIfdefelif _ ->
let body, extras, xs = mk_ifdef_parameters (x::extras) [] xs in
(List.rev acc_before_sep)::body, extras, xs
| _ ->
let line, xs = Common.span (fun y -> y.line = x.line) (x::xs) in
mk_ifdef_parameters extras (NotIfdefLine line::acc_before_sep) xs
)
(* ------------------------------------------------------------------------- *)
(* Lines (of parens) *)
(* ------------------------------------------------------------------------- *)
let line_of_paren = function
| PToken x -> x.line
| Parenthised (_xxs, info_parens) ->
(match info_parens with
| [] -> raise Impossible
| x::_xs -> x.line
)
(* old
let rec span_line_paren line = function
| [] -> [],[]
| x::xs ->
(match x with
| PToken tok when TH.is_eof tok.t ->
[], x::xs
| _ ->
if line_of_paren x = line
then
let (l1, l2) = span_line_paren line xs in
(x::l1, l2)
else ([], x::xs)
)
let rec mk_line_parenthised xs =
match xs with
| [] -> []
| x::xs ->
let line_no = line_of_paren x in
let line, xs = span_line_paren line_no xs in
Line (x::line)::mk_line_parenthised xs
*)
let line_range_of_paren = function
| PToken x -> x.line, x.line
| Parenthised (_xxs, info_parens) ->
(match info_parens with
| [] -> raise Impossible
| x::xs ->
let lines_no = (x::xs) +> List.map (fun x -> x.line) in
Common2.minimum lines_no, Common2.maximum lines_no
)
let rec span_line_paren_range (imin, imax) = function
| [] -> [],[]
| x::xs ->
(match x with
| PToken tok when TH.is_eof tok.t ->
[], x::xs
| _ ->
if line_of_paren x >= imin && line_of_paren x <= imax
then
(* may need to extend *)
let (_imin', imax') = line_range_of_paren x in
let (l1, l2) = span_line_paren_range (imin, max imax imax') xs in
(x::l1, l2)
else ([], x::xs)
)
let rec mk_line_parenthised xs =
match xs with
| [] -> []
| x::xs ->
let line_range = line_range_of_paren x in
let line, xs = span_line_paren_range line_range xs in
Line (x::line)::mk_line_parenthised xs
(* ------------------------------------------------------------------------- *)
(* Function body *)
(* ------------------------------------------------------------------------- *)
let rec mk_body_function_grouped xs =
match xs with
| [] -> []
| x::xs ->
(match x with
| {t=TOBrace _; col = 0; _} ->
let is_closing_brace = function
| {t = TCBrace _; col = 0; _ } -> true
| _ -> false
in
let body, xs = Common.span (fun x -> not (is_closing_brace x)) xs in
(match xs with
| ({t = TCBrace _; col = 0; _ })::xs ->
BodyFunction body::mk_body_function_grouped xs
| [] ->
pr2 "PB:not found closing brace in fuzzy parsing";
[NotBodyLine body]
| _ -> raise Impossible
)
| _ ->
let line, xs = Common.span (fun y -> y.line = x.line) (x::xs) in
NotBodyLine line::mk_body_function_grouped xs
)
(* ------------------------------------------------------------------------- *)
(* Multi ('{', '(', '<') (could also do '[' ?) *)
(* ------------------------------------------------------------------------- *)
(* Assumes work on a list of tokens without comments, without ifdefs
* (todo? and without #define?).
* Used for typedef inference. Now also used for fuzzy parsing!
*
* todo? more fault tolerance, if col == 0 and { the reset!
* less: could check that it's consistent with the indentation
*
*)
let mk_multi xs =
let rec consume x xs =
match x with
| {t=(*TOBrace ii*)tok;_} when TH.is_obrace tok ->
let body, closing, rest = look_close_brace x [] xs in
Braces (x, body, closing), rest
| {t=(*TOPar ii*)tok;_} when TH.is_opar tok ->
let body, closing, rest = look_close_paren x [] xs in
Parens (x, body, closing), rest
| {t=TInf_Template _ii;_} ->
let body, closing, rest = look_close_template x [] xs in
Angle (x, body, closing), rest
| x -> Tok x, xs
and aux xs =
match xs with
| [] -> []
| x::xs ->
let x', xs' = consume x xs in
x'::aux xs'
and look_close_brace tok_start accbody xs =
match xs with
| [] ->
raise (UnclosedSymbol (spf "PB look_close_brace (started at %d)"
(TH.line_of_tok tok_start.t)))
| x::xs ->
(match x with
| {t=TCBrace _ii;_} -> List.rev accbody, Some x, xs
(* Many macros have unclosed '{'. An alternative
* would be to work on a view where define has been filtered
*)
| {t=TCommentNewline_DefineEndOfMacro _ii;_} ->
List.rev accbody, None, x::xs
| _ -> let (x', xs') = consume x xs in
look_close_brace tok_start (x'::accbody) xs'
)
and look_close_paren tok_start accbody xs =
match xs with
| [] ->
raise (UnclosedSymbol (spf "PB look_close_paren (started at %d)"
(TH.line_of_tok tok_start.t)))
| x::xs ->
(match x with
| {t=(*TCPar ii*)tok;_} when TH.is_cpar tok ->
List.rev accbody, Some x, xs
| _ ->
let (x', xs') = consume x xs in
look_close_paren tok_start (x'::accbody) xs'
)
and look_close_template tok_start accbody xs =
match xs with
| [] ->
raise (UnclosedSymbol (spf "PB look_close_template (started at %d)"
(TH.line_of_tok tok_start.t)))
| x::xs ->
(match x with
| {t=TSup_Template _ii;_} -> List.rev accbody, Some x, xs
| _ -> let (x', xs') = consume x xs in
look_close_template tok_start (x'::accbody) xs'
)
in
aux xs
let split_comma xs =
xs +> Common2.split_gen_when (function
| Tok{t=TComma _;_}::xs -> Some xs
| _ -> None
)
(*****************************************************************************)
(* View iterators *)
(*****************************************************************************)
let rec iter_token_paren f xs =
xs +> List.iter (function
| PToken tok -> f tok;
| Parenthised (xxs, info_parens) ->
info_parens +> List.iter f;
xxs +> List.iter (fun xs -> iter_token_paren f xs)
)
let rec iter_token_brace f xs =
xs +> List.iter (function
| BToken tok -> f tok;
| Braceised (xxs, tok1, tok2opt) ->
f tok1; do_option f tok2opt;
xxs +> List.iter (fun xs -> iter_token_brace f xs)
)
let rec iter_token_ifdef f xs =
xs +> List.iter (function
| NotIfdefLine xs -> xs +> List.iter f;
| Ifdefbool (_, xxs, info_ifdef)
| Ifdef (xxs, info_ifdef) ->
info_ifdef +> List.iter f;
xxs +> List.iter (iter_token_ifdef f)
)
let rec iter_token_multi f xs =
xs +> List.iter (function
| Tok t -> f t
| Braces (t1, xs, t2)
| Parens (t1, xs, t2)
| Angle (t1, xs, t2)
->
f t1;
iter_token_multi f xs;
Common.do_option f t2
)
let tokens_of_paren xs =
let g = ref [] in
xs +> iter_token_paren (fun tok -> push tok g);
List.rev !g
let tokens_of_paren_ordered xs =
let g = ref [] in
let rec aux_tokens_ordered = function
| PToken tok -> push tok g;
| Parenthised (xxs, info_parens) ->
let (opar, cpar, commas) =
match info_parens with
| opar::xs ->
(match List.rev xs with
| cpar::xs ->
opar, cpar, List.rev xs
| _ -> raise Impossible
)
| _ -> raise Impossible
in
push opar g;
aux_args (xxs,commas);
push cpar g;
and aux_args (xxs, commas) =
match xxs, commas with
| [], [] -> ()
| [xs], [] -> xs +> List.iter aux_tokens_ordered
| xs::ys::xxs, comma::commas ->
xs +> List.iter aux_tokens_ordered;
push comma g;
aux_args (ys::xxs, commas)
| _ -> raise Impossible
in
xs +> List.iter aux_tokens_ordered;
List.rev !g
let tokens_of_multi_grouped xs =
let res = ref [] in
let add x = Common.push x res in
let rec aux xs =
xs +> List.iter (function
| Tok t1 -> add t1
| Braces (t1, xs, t2)
| Parens (t1, xs, t2)
| Angle (t1, xs, t2) ->
add t1;
aux xs;
Common.do_option add t2
)
in
aux xs;
List.rev !res
(*****************************************************************************)
(* vof *)
(*****************************************************************************)
let vof_context = function
| InTopLevel -> Ocaml.VSum ("T", [])
| InClassStruct _s -> Ocaml.VSum ("C", [])
| InEnum -> Ocaml.VSum ("E", [])
| InInitializer -> Ocaml.VSum ("I", [])
| InAssign -> Ocaml.VSum ("=", [])
| InParameter -> Ocaml.VSum ("P", [])
| InArgument -> Ocaml.VSum ("A", [])
| InFunction -> Ocaml.VSum ("F", [])
(*
| InTemplateParam -> Ocaml.VSum ("<>", [])
*)
let vof_token_extended t =
let info = TH.info_of_tok t.t in
let str = PI.str_of_info info in
let xs = List.map vof_context t.where in
Ocaml.VTuple [Ocaml.VString str; Ocaml.VList xs]
let rec vof_multi_grouped =
function
| Braces ((v1, v2, v3)) ->
let v1 = vof_token_extended v1
and v2 = Ocaml.vof_list vof_multi_grouped v2
and v3 = Ocaml.vof_option vof_token_extended v3
in Ocaml.VSum (("Braces", [ v1; v2; v3 ]))
| Parens ((v1, v2, v3)) ->
let v1 = vof_token_extended v1
and v2 = Ocaml.vof_list vof_multi_grouped v2
and v3 = Ocaml.vof_option vof_token_extended v3
in Ocaml.VSum (("Parens", [ v1; v2; v3 ]))
| Angle ((v1, v2, v3)) ->
let v1 = vof_token_extended v1
and v2 = Ocaml.vof_list vof_multi_grouped v2
and v3 = Ocaml.vof_option vof_token_extended v3
in Ocaml.VSum (("Angle", [ v1; v2; v3 ]))
| Tok v1 -> let v1 = vof_token_extended v1 in Ocaml.VSum (("Tok", [ v1 ]))
let vof_multi_grouped_list xs =
let v = Ocaml.VList (xs +> List.map vof_multi_grouped) in
v

View file

@ -0,0 +1,72 @@
type token_extended = {
mutable t: Parser_cpp.token;
mutable where : context list;
mutable new_tokens_before : Parser_cpp.token list;
line : int;
col : int;
}
and context =
| InTopLevel
| InClassStruct of string
| InEnum
| InInitializer
| InAssign
| InParameter
| InArgument
| InFunction
(*
| InTemplateParam
*)
val mk_token_extended : Parser_cpp.token -> token_extended
val mk_token_fake : Parser_cpp.token -> token_extended
val rebuild_tokens_extented : token_extended list -> token_extended list
type paren_grouped =
| Parenthised of paren_grouped list list * token_extended list
| PToken of token_extended
type brace_grouped =
| Braceised of brace_grouped list list * token_extended *
token_extended option
| BToken of token_extended
type ifdef_grouped =
| Ifdef of ifdef_grouped list list * token_extended list
| Ifdefbool of bool * ifdef_grouped list list * token_extended list
| NotIfdefLine of token_extended list
type 'a line_grouped =
Line of 'a list
type body_function_grouped =
| BodyFunction of token_extended list
| NotBodyLine of token_extended list
type multi_grouped =
| Braces of token_extended * multi_grouped list * token_extended option
| Parens of token_extended * multi_grouped list * token_extended option
| Angle of token_extended * multi_grouped list * token_extended option
| Tok of token_extended
val split_comma: multi_grouped list -> multi_grouped list list
val mk_parenthised: token_extended list -> paren_grouped list
val mk_braceised: token_extended list -> brace_grouped list
val mk_ifdef: token_extended list -> ifdef_grouped list
val mk_body_function_grouped: token_extended list -> body_function_grouped list
val mk_line_parenthised: paren_grouped list -> paren_grouped line_grouped list
exception UnclosedSymbol of string
val mk_multi: token_extended list -> multi_grouped list
val iter_token_paren : (token_extended -> unit) -> paren_grouped list -> unit
val iter_token_brace : (token_extended -> unit) -> brace_grouped list -> unit
val iter_token_ifdef : (token_extended -> unit) -> ifdef_grouped list -> unit
val iter_token_multi : (token_extended -> unit) -> multi_grouped list -> unit
val tokens_of_paren: paren_grouped list -> token_extended list
val tokens_of_paren_ordered: paren_grouped list -> token_extended list
val tokens_of_multi_grouped: multi_grouped list -> token_extended list
val vof_multi_grouped_list: multi_grouped list -> Ocaml.v

View file

@ -0,0 +1,42 @@
(* Yoann Padioleau
*
* Copyright (C) 2010 Facebook
*
* This library is free software; you can redistribute it and/or
* modify it under the terms of the GNU Lesser General Public License
* version 2.1 as published by the Free Software Foundation, with the
* special exception on linking described in file license.txt.
*
* This library is distributed in the hope that it will be useful, but
* WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the file
* license.txt for more details.
*)
open Ast_cpp
module Ast = Ast_cpp
(*****************************************************************************)
(* Prelude *)
(*****************************************************************************)
(*****************************************************************************)
(* Accessors *)
(*****************************************************************************)
let is_function_type x =
match Ast.unwrap_typeC x with
| FunctionType _ -> true
| _ -> false
let rec is_method_type x =
match Ast.unwrap_typeC x with
| Pointer y ->
is_method_type y
| ParenType paren_ft ->
is_method_type (Ast.unparen paren_ft)
| FunctionType _ ->
true
| _ -> false

View file

@ -0,0 +1,4 @@
val is_function_type: Ast_cpp.fullType -> bool
val is_method_type: Ast_cpp.fullType -> bool

View file

@ -0,0 +1,80 @@
open Common
open OUnit
module Ast = Ast_cpp
module Flag = Flag_parsing_cpp
(*****************************************************************************)
(* Helpers *)
(*****************************************************************************)
let parse file =
Common.save_excursion Flag.error_recovery false (fun () ->
Common.save_excursion Flag.show_parsing_error false (fun () ->
Common.save_excursion Flag.verbose_parsing false (fun () ->
Parse_cpp.parse file
)))
(*****************************************************************************)
(* Unit tests *)
(*****************************************************************************)
let unittest =
"parsing_cpp" >::: [
(*-----------------------------------------------------------------------*)
(* Lexing *)
(*-----------------------------------------------------------------------*)
(* todo:
* - make sure parse int correctly, and float, and that actually does
* not return multiple tokens for 42.42
* - ...
*)
(*-----------------------------------------------------------------------*)
(* Parsing *)
(*-----------------------------------------------------------------------*)
"regression files" >:: (fun () ->
let dir = Filename.concat Config_pfff.path "/tests/cpp/parsing" in
let files =
Common2.glob (spf "%s/*.cpp" dir) @ Common2.glob (spf "%s/*.h" dir) in
files +> List.iter (fun file ->
try
let _ast = parse file in
()
with Parse_cpp.Parse_error _ ->
assert_failure (spf "it should correctly parse %s" file)
)
);
"rejecting bad code" >:: (fun () ->
let dir = Filename.concat Config_pfff.path "/tests/cpp/parsing_errors" in
let files = Common2.glob (spf "%s/*.cpp" dir) in
files +> List.iter (fun file ->
try
let _ast = parse file in
assert_failure (spf "it should have thrown a Parse_error %s" file)
with
| Parse_cpp.Parse_error _ -> ()
| exn -> assert_failure (spf "throwing wrong exn %s on %s"
(Common.exn_to_s exn) file)
)
);
(* parsing C files (and not C++ files) possibly containing C++ keywords *)
"C regression files" >:: (fun () ->
let dir = Filename.concat Config_pfff.path "/tests/c/parsing" in
let files =
Common2.glob (spf "%s/*.c" dir)
(* @ Common2.glob (spf "%s/*.h" dir) *) in
files +> List.iter (fun file ->
try
let _ast = parse file in
()
with Parse_cpp.Parse_error _ ->
assert_failure (spf "it should correctly parse %s" file)
)
);
(*-----------------------------------------------------------------------*)
(* Misc *)
(*-----------------------------------------------------------------------*)
]

View file

@ -0,0 +1,5 @@
(* Returns the testsuite for this directory. To be concatenated by
* the caller (e.g. in pfff/main_test.ml ) with other testsuites and
* run via OUnit.run_test_tt
*)
val unittest: OUnit.test

View file

@ -0,0 +1,879 @@
(* Yoann Padioleau
*
* Copyright (C) 2010 Facebook
*
* This library is free software; you can redistribute it and/or
* modify it under the terms of the GNU Lesser General Public License
* version 2.1 as published by the Free Software Foundation, with the
* special exception on linking described in file license.txt.
*
* This library is distributed in the hope that it will be useful, but
* WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the file
* license.txt for more details.
*)
open Ocaml
open Ast_cpp
(*****************************************************************************)
(* Prelude *)
(*****************************************************************************)
(*****************************************************************************)
(* Types *)
(*****************************************************************************)
(* hooks *)
type visitor_in = {
kexpr: expression vin;
kstmt: statement vin;
kinit: initialiser vin;
ktypeC: typeC vin;
kclass_member: class_member vin;
kfieldkind: fieldkind vin;
kparameter: parameter vin;
kcompound: compound vin;
kclass_def: class_definition vin;
kfunc_def: func_definition vin;
kcpp: cpp_directive vin;
kblock_decl: block_declaration vin;
kdeclaration: declaration vin;
ktoplevel: toplevel vin;
kinfo: tok vin;
}
and visitor_out = any -> unit
and 'a vin = ('a -> unit) * visitor_out -> 'a -> unit
let default_visitor =
{ kexpr = (fun (k,_) x -> k x);
kfieldkind = (fun (k,_) x -> k x);
kparameter = (fun (k,_) x -> k x);
ktypeC = (fun (k,_) x -> k x);
kblock_decl = (fun (k,_) x -> k x);
kcompound = (fun (k,_) x -> k x);
kstmt = (fun (k,_) x -> k x);
kinfo = (fun (k,_) x -> k x);
kclass_def = (fun (k,_) x -> k x);
kfunc_def = (fun (k,_) x -> k x);
kclass_member = (fun (k,_) x -> k x);
kcpp = (fun (k,_) x -> k x);
kdeclaration = (fun (k,_) x -> k x);
ktoplevel = (fun (k,_) x -> k x);
kinit = (fun (k,_) x -> k x);
}
let (mk_visitor: visitor_in -> visitor_out) = fun vin ->
(* start of auto generation *)
(* generated by ocamltarzan with: camlp4o -o /tmp/yyy.ml -I pa/ pa_type_conv.cmo pa_visitor.cmo pr_o.cmo /tmp/xxx.ml *)
let rec v_info x =
let k _ = () in
vin.kinfo (k, all_functions) x
and v_tok v = v_info v
and v_wrap:'a. ('a -> unit) -> 'a wrap -> unit =
fun _of_a (v1, v2) ->
let v1 = _of_a v1 and v2 = v_list v_info v2 in ()
and v_wrap2:'a. ('a -> unit) -> 'a wrap2 -> unit =
fun _of_a (v1, v2) ->
let v1 = _of_a v1 and v2 = v_info v2 in ()
and v_paren:'a. ('a -> unit) -> 'a paren -> unit =
fun _of_a (v1, v2, v3) ->
let v1 = v_tok v1 and v2 = _of_a v2 and v3 = v_tok v3 in ()
and v_brace: 'a. ('a -> unit) -> 'a brace -> unit =
fun _of_a (v1, v2, v3) ->
let v1 = v_tok v1 and v2 = _of_a v2 and v3 = v_tok v3 in ()
and v_bracket: 'a. ('a -> unit) -> 'a bracket -> unit =
fun _of_a (v1, v2, v3) ->
let v1 = v_tok v1 and v2 = _of_a v2 and v3 = v_tok v3 in ()
and v_angle: 'a. ('a -> unit) -> 'a angle -> unit =
fun _of_a (v1, v2, v3) ->
let v1 = v_tok v1 and v2 = _of_a v2 and v3 = v_tok v3 in ()
and v_comma_list: 'a. ('a -> unit) -> 'a comma_list -> unit = fun
_of_a -> v_list (v_wrap _of_a)
and v_comma_list2: 'a. ('a -> unit) -> 'a comma_list2 -> unit =
fun _of_a ->
v_list (Ocaml.v_either _of_a v_tok)
and v_name (v1, v2, v3) =
let v1 = v_option v_tok v1
and v2 =
v_list (fun (v1, v2) -> let v1 = v_qualifier v1 and v2 = v_tok v2 in ())
v2
and v3 = v_ident v3
in ()
and v_ident =
function
| IdIdent v1 -> let v1 = v_wrap2 v_string v1 in ()
| IdOperator ((v1, v2)) ->
let v1 = v_tok v1
and v2 =
(match v2 with
| (v1, v2) -> let v1 = v_operator v1 and v2 = v_list v_tok v2 in ())
in ()
| IdConverter ((v1, v2)) -> let v1 = v_tok v1 and v2 = v_fullType v2 in ()
| IdDestructor ((v1, v2)) ->
let v1 = v_tok v1 and v2 = v_wrap2 v_string v2 in ()
| IdTemplateId ((v1, v2)) ->
let v1 = v_wrap2 v_string v1 and v2 = v_template_arguments v2 in ()
and v_template_arguments v = v_angle (v_comma_list v_template_argument) v
and v_template_argument v = Ocaml.v_either v_fullType v_expression v
and v_either_ft_or_expr v = Ocaml.v_either v_fullType v_expression v
and v_qualifier =
function
| QClassname v1 -> let v1 = v_wrap2 v_string v1 in ()
| QTemplateId ((v1, v2)) ->
let v1 = v_wrap2 v_string v1 and v2 = v_template_arguments v2 in ()
and v_class_name v = v_name v
and v_namespace_name v = v_name v
and v_typedef_name v = v_name v
and v_enum_name v = v_name v
and v_ident_name v = v_name v
and v_fullType (v1, v2) =
let v1 = v_typeQualifier v1 and v2 = v_typeC v2 in ()
and v_typeC v =
let k v = v_wrap v_typeCbis v in
vin.ktypeC (k, all_functions) v
and v_typeCbis =
function
| BaseType v1 -> let v1 = v_baseType v1 in ()
| Pointer v1 -> let v1 = v_fullType v1 in ()
| Reference v1 -> let v1 = v_fullType v1 in ()
| Array ((v1, v2)) ->
let v1 = v_bracket (v_option v_constExpression) v1
and v2 = v_fullType v2
in ()
| FunctionType v1 -> let v1 = v_functionType v1 in ()
| EnumDef ((v1, v2, v3)) ->
let v1 = v_tok v1
and v2 = v_option (v_wrap2 v_string) v2
and v3 = v_brace (v_comma_list v_enum_elem) v3
in ()
| StructDef v1 -> let v1 = v_class_definition v1 in ()
| EnumName ((v1, v2)) ->
let v1 = v_tok v1 and v2 = v_wrap2 v_string v2 in ()
| StructUnionName ((v1, v2)) ->
let v1 = v_wrap2 v_structUnion v1 and v2 = v_wrap2 v_string v2 in ()
| TypeName ((v1)) ->
let v1 = v_name v1 in ()
| TypenameKwd ((v1, v2)) -> let v1 = v_tok v1 and v2 = v_name v2 in ()
| TypeOf ((v1, v2)) ->
let v1 = v_tok v1 and v2 = v_paren v_either_ft_or_expr v2 in ()
| ParenType v1 -> let v1 = v_paren v_fullType v1 in ()
and v_baseType =
function
| Void -> ()
| IntType v1 -> let v1 = v_intType v1 in ()
| FloatType v1 -> let v1 = v_floatType v1 in ()
and v_intType =
function
| CChar -> ()
| Si v1 -> let v1 = v_signed v1 in ()
| CBool -> ()
| WChar_t -> ()
and v_signed (v1, v2) = let v1 = v_sign v1 and v2 = v_base v2 in ()
and v_base =
function
| CChar2 -> ()
| CShort -> ()
| CInt -> ()
| CLong -> ()
| CLongLong -> ()
and v_sign = function | Signed -> () | UnSigned -> ()
and v_floatType = function | CFloat -> () | CDouble -> () | CLongDouble -> ()
and v_enum_elem { e_name = v_e_name; e_val = v_e_val } =
let arg = v_wrap2 v_string v_e_name in
let arg =
v_option
(fun (v1, v2) -> let v1 = v_tok v1 and v2 = v_constExpression v2 in ())
v_e_val
in ()
and v_typeQualifier { const = v_const; volatile = v_volatile } =
let arg = v_option v_tok v_const in
let arg = v_option v_tok v_volatile in ()
and v_expression v =
let k x = v_wrap v_expressionbis x in
vin.kexpr (k, all_functions) v
and v_expressionbis =
function
| Id ((v1, v2)) -> let v1 = v_name v1 and v2 = v_ident_info v2 in ()
| C v1 -> let v1 = v_constant v1 in ()
| Call ((v1, v2)) ->
let v1 = v_expression v1
and v2 = v_paren (v_comma_list v_argument) v2
in ()
| CondExpr ((v1, v2, v3)) ->
let v1 = v_expression v1
and v2 = v_option v_expression v2
and v3 = v_expression v3
in ()
| Sequence ((v1, v2)) ->
let v1 = v_expression v1 and v2 = v_expression v2 in ()
| Assignment ((v1, v2, v3)) ->
let v1 = v_expression v1
and v2 = v_assignOp v2
and v3 = v_expression v3
in ()
| Postfix ((v1, v2)) -> let v1 = v_expression v1 and v2 = v_fixOp v2 in ()
| Infix ((v1, v2)) -> let v1 = v_expression v1 and v2 = v_fixOp v2 in ()
| Unary ((v1, v2)) -> let v1 = v_expression v1 and v2 = v_unaryOp v2 in ()
| Binary ((v1, v2, v3)) ->
let v1 = v_expression v1
and v2 = v_binaryOp v2
and v3 = v_expression v3
in ()
| ArrayAccess ((v1, v2)) ->
let v1 = v_expression v1 and v2 = v_bracket v_expression v2 in ()
| RecordAccess ((v1, v2)) ->
let v1 = v_expression v1 and v2 = v_name v2 in ()
| RecordPtAccess ((v1, v2)) ->
let v1 = v_expression v1 and v2 = v_name v2 in ()
| RecordStarAccess ((v1, v2)) ->
let v1 = v_expression v1 and v2 = v_expression v2 in ()
| RecordPtStarAccess ((v1, v2)) ->
let v1 = v_expression v1 and v2 = v_expression v2 in ()
| SizeOfExpr ((v1, v2)) -> let v1 = v_tok v1 and v2 = v_expression v2 in ()
| SizeOfType ((v1, v2)) ->
let v1 = v_tok v1 and v2 = v_paren v_fullType v2 in ()
| Cast ((v1, v2)) ->
let v1 = v_paren v_fullType v1 and v2 = v_expression v2 in ()
| StatementExpr v1 -> let v1 = v_paren v_compound v1 in ()
| GccConstructor ((v1, v2)) ->
let v1 = v_paren v_fullType v1
and v2 = v_brace (v_comma_list v_initialiser) v2
in ()
| This v1 -> let v1 = v_tok v1 in ()
| ConstructedObject ((v1, v2)) ->
let v1 = v_fullType v1
and v2 = v_paren (v_comma_list v_argument) v2
in ()
| TypeId ((v1, v2)) ->
let v1 = v_tok v1 and v2 = v_paren v_either_ft_or_expr v2 in ()
| CplusplusCast ((v1, v2, v3)) ->
let v1 = v_wrap2 v_cast_operator v1
and v2 = v_angle v_fullType v2
and v3 = v_paren v_expression v3
in ()
| New ((v1, v2, v3, v4, v5)) ->
let v1 = v_option v_tok v1
and v2 = v_tok v2
and v3 = v_option (v_paren (v_comma_list v_argument)) v3
and v4 = v_fullType v4
and v5 = v_option (v_paren (v_comma_list v_argument)) v5
in ()
| Delete ((v1, v2)) ->
let v1 = v_option v_tok v1 and v2 = v_expression v2 in ()
| DeleteArray ((v1, v2)) ->
let v1 = v_option v_tok v1 and v2 = v_expression v2 in ()
| Throw v1 -> let v1 = v_option v_expression v1 in ()
| ParenExpr v1 -> let v1 = v_paren v_expression v1 in ()
| ExprTodo -> ()
and v_ident_info { i_scope = _v_i_scope } =
(* todo? let arg = Scope_code.v_scope v_i_scope in () *)
()
and v_argument v = Ocaml.v_either v_expression v_weird_argument v
and v_weird_argument =
function
| ArgType v1 -> let v1 = v_fullType v1 in ()
| ArgAction v1 -> let v1 = v_action_macro v1 in ()
and v_action_macro = function | ActMisc v1 -> let v1 = v_list v_tok v1 in ()
and v_constant =
function
| String v1 ->
let v1 =
(match v1 with
| (v1, v2) -> let v1 = v_string v1 and v2 = v_isWchar v2 in ())
in ()
| MultiString -> ()
| Char v1 ->
let v1 =
(match v1 with
| (v1, v2) -> let v1 = v_string v1 and v2 = v_isWchar v2 in ())
in ()
| Int v1 -> let v1 = v_string v1 in ()
| Float v1 ->
let v1 =
(match v1 with
| (v1, v2) -> let v1 = v_string v1 and v2 = v_floatType v2 in ())
in ()
| Bool v1 -> let v1 = v_bool v1 in ()
and v_isWchar = function | IsWchar -> () | IsChar -> ()
and v_unaryOp =
function
| GetRef -> ()
| DeRef -> ()
| UnPlus -> ()
| UnMinus -> ()
| Tilde -> ()
| Not -> ()
| GetRefLabel -> ()
and v_assignOp =
function | SimpleAssign -> () | OpAssign v1 -> let v1 = v_arithOp v1 in ()
and v_fixOp = function | Dec -> () | Inc -> ()
and v_binaryOp =
function
| Arith v1 -> let v1 = v_arithOp v1 in ()
| Logical v1 -> let v1 = v_logicalOp v1 in ()
and v_arithOp =
function
| Plus -> ()
| Minus -> ()
| Mul -> ()
| Div -> ()
| Mod -> ()
| DecLeft -> ()
| DecRight -> ()
| And -> ()
| Or -> ()
| Xor -> ()
and v_logicalOp =
function
| Inf -> ()
| Sup -> ()
| InfEq -> ()
| SupEq -> ()
| Eq -> ()
| NotEq -> ()
| AndLog -> ()
| OrLog -> ()
and v_ptrOp = function | PtrStarOp -> () | PtrOp -> ()
and v_allocOp =
function
| NewOp -> ()
| DeleteOp -> ()
| NewArrayOp -> ()
| DeleteArrayOp -> ()
and v_accessop = function | ParenOp -> () | ArrayOp -> ()
and v_operator =
function
| BinaryOp v1 -> let v1 = v_binaryOp v1 in ()
| AssignOp v1 -> let v1 = v_assignOp v1 in ()
| FixOp v1 -> let v1 = v_fixOp v1 in ()
| PtrOpOp v1 -> let v1 = v_ptrOp v1 in ()
| AccessOp v1 -> let v1 = v_accessop v1 in ()
| AllocOp v1 -> let v1 = v_allocOp v1 in ()
| UnaryTildeOp -> ()
| UnaryNotOp -> ()
| CommaOp -> ()
and v_cast_operator =
function
| Static_cast -> ()
| Dynamic_cast -> ()
| Const_cast -> ()
| Reinterpret_cast -> ()
and v_constExpression v = v_expression v
and v_statement v =
let k v = v_wrap v_statementbis v in
vin.kstmt (k, all_functions) v
and v_statementbis =
function
| Compound v1 -> let v1 = v_compound v1 in ()
| ExprStatement v1 -> let v1 = v_exprStatement v1 in ()
| Labeled v1 -> let v1 = v_labeled v1 in ()
| Selection v1 -> let v1 = v_selection v1 in ()
| Iteration v1 -> let v1 = v_iteration v1 in ()
| Jump v1 -> let v1 = v_jump v1 in ()
| DeclStmt v1 -> let v1 = v_block_declaration v1 in ()
| Try ((v1, v2, v3)) ->
let v1 = v_tok v1
and v2 = v_compound v2
and v3 = v_list v_handler v3
in ()
| NestedFunc v1 -> let v1 = v_func_definition v1 in ()
| MacroStmt -> ()
| StmtTodo -> ()
and v_compound v =
let k v = v_brace (v_list v_statement_sequencable) v in
vin.kcompound (k, all_functions) v
and v_statement_sequencable =
function
| StmtElem v1 -> let v1 = v_statement v1 in ()
| CppDirectiveStmt v1 -> let v1 = v_cpp_directive v1 in ()
| IfdefStmt v1 -> let v1 = v_ifdef_directive v1 in ()
and v_exprStatement v = v_option v_expression v
and v_labeled =
function
| Label ((v1, v2)) -> let v1 = v_string v1 and v2 = v_statement v2 in ()
| Case ((v1, v2)) -> let v1 = v_expression v1 and v2 = v_statement v2 in ()
| CaseRange ((v1, v2, v3)) ->
let v1 = v_expression v1
and v2 = v_expression v2
and v3 = v_statement v3
in ()
| Default v1 -> let v1 = v_statement v1 in ()
and v_selection =
function
| If ((v1, v2, v3, v4, v5)) ->
let v1 = v_tok v1
and v2 = v_paren v_expression v2
and v3 = v_statement v3
and v4 = v_option v_tok v4
and v5 = v_statement v5
in ()
| Switch ((v1, v2, v3)) ->
let v1 = v_tok v1
and v2 = v_paren v_expression v2
and v3 = v_statement v3
in ()
and v_iteration =
function
| While ((v1, v2, v3)) ->
let v1 = v_tok v1
and v2 = v_paren v_expression v2
and v3 = v_statement v3
in ()
| DoWhile ((v1, v2, v3, v4, v5)) ->
let v1 = v_tok v1
and v2 = v_statement v2
and v3 = v_tok v3
and v4 = v_paren v_expression v4
and v5 = v_tok v5
in ()
| For ((v1, v2, v3)) ->
let v1 = v_tok v1
and v2 =
v_paren
(fun (v1, v2, v3) ->
let v1 = v_wrap v_exprStatement v1
and v2 = v_wrap v_exprStatement v2
and v3 = v_wrap v_exprStatement v3
in ())
v2
and v3 = v_statement v3
in ()
| MacroIteration ((v1, v2, v3)) ->
let v1 = v_wrap2 v_string v1
and v2 = v_paren (v_comma_list v_argument) v2
and v3 = v_statement v3
in ()
and v_jump =
function
| Goto v1 -> let v1 = v_string v1 in ()
| Continue -> ()
| Break -> ()
| Return -> ()
| ReturnExpr v1 -> let v1 = v_expression v1 in ()
| GotoComputed v1 -> let v1 = v_expression v1 in ()
and v_handler (v1, v2, v3) =
let v1 = v_tok v1
and v2 = v_paren v_exception_declaration v2
and v3 = v_compound v3
in ()
and v_exception_declaration =
function
| ExnDeclEllipsis v1 -> let v1 = v_tok v1 in ()
| ExnDecl v1 -> let v1 = v_parameter v1 in ()
and v_block_declaration x =
let k = function
| DeclList ((v1, v2)) ->
let v1 = v_comma_list v_onedecl v1 and v2 = v_tok v2 in ()
| MacroDecl ((v1, v2, v3, v4)) ->
let v1 = v_list v_tok v1
and v2 = v_wrap2 v_string v2
and v3 = v_paren (v_comma_list v_argument) v3
and v4 = v_tok v4
in ()
| UsingDecl v1 ->
let v1 =
(match v1 with
| (v1, v2, v3) ->
let v1 = v_tok v1 and v2 = v_name v2 and v3 = v_tok v3 in ())
in ()
| UsingDirective ((v1, v2, v3, v4)) ->
let v1 = v_tok v1
and v2 = v_tok v2
and v3 = v_namespace_name v3
and v4 = v_tok v4
in ()
| NameSpaceAlias ((v1, v2, v3, v4, v5)) ->
let v1 = v_tok v1
and v2 = v_wrap2 v_string v2
and v3 = v_tok v3
and v4 = v_namespace_name v4
and v5 = v_tok v5
in ()
| Asm ((v1, v2, v3, v4)) ->
let v1 = v_tok v1
and v2 = v_option v_tok v2
and v3 = v_paren v_asmbody v3
and v4 = v_tok v4
in ()
in
vin.kblock_decl (k, all_functions) x
and
v_onedecl { v_namei = v_v_namei; v_type = v_v_type; v_storage = v_v_storage
} =
let arg =
v_option
(fun (v1, v2) -> let v1 = v_name v1 and v2 = v_option v_init v2 in ())
v_v_namei in
let arg = v_fullType v_v_type in
let arg = v_storage v_v_storage in ()
and v_storage v = v_storagebis v
and v_storagebis =
function
| NoSto -> ()
| StoTypedef v1 -> v_tok v1
| Sto v1 -> let v1 = v_wrap2 v_storageClass v1 in ()
and v_storageClass =
function | Auto -> () | Static -> () | Register -> () | Extern -> ()
and v_func_specifier = function | Inline -> () | Virtual -> ()
and v_init =
function
| EqInit ((v1, v2)) -> let v1 = v_tok v1 and v2 = v_initialiser v2 in ()
| ObjInit v1 -> let v1 = v_paren (v_comma_list v_argument) v1 in ()
and v_initialiser x =
let k x =
match x with
| InitExpr v1 -> let v1 = v_expression v1 in ()
| InitList v1 -> let v1 = v_brace (v_comma_list v_initialiser) v1 in ()
| InitDesignators ((v1, v2, v3)) ->
let v1 = v_list v_designator v1
and v2 = v_tok v2
and v3 = v_initialiser v3
in ()
| InitFieldOld ((v1, v2, v3)) ->
let v1 = v_wrap2 v_string v1
and v2 = v_tok v2
and v3 = v_initialiser v3
in ()
| InitIndexOld ((v1, v2)) ->
let v1 = v_bracket v_expression v1 and v2 = v_initialiser v2 in ()
in
vin.kinit (k, all_functions) x
and v_designator =
function
| DesignatorField ((v1, v2)) ->
let v1 = v_tok v1 and v2 = v_wrap2 v_string v2 in ()
| DesignatorIndex v1 -> let v1 = v_bracket v_expression v1 in ()
| DesignatorRange v1 ->
let v1 =
v_bracket
(fun (v1, v2, v3) ->
let v1 = v_expression v1
and v2 = v_tok v2
and v3 = v_expression v3
in ())
v1
in ()
and v_asmbody (v1, v2) =
let v1 = v_list v_tok v1 and v2 = v_list (v_wrap v_colon) v2 in ()
and v_colon =
function | Colon v1 -> let v1 = v_comma_list v_colon_option v1 in ()
and v_colon_option v = v_wrap v_colon_optionbis v
and v_colon_optionbis =
function
| ColonMisc -> ()
| ColonExpr v1 -> let v1 = v_paren v_expression v1 in ()
and
v_func_definition x =
let k = function {
f_name = v_f_name;
f_type = v_f_type;
f_storage = v_f_storage;
f_body = v_f_body
} ->
let arg = v_name v_f_name in
let arg = v_functionType v_f_type in
let arg = v_storage v_f_storage in
let arg = v_compound v_f_body in ()
in
vin.kfunc_def (k, all_functions) x
and
v_functionType {
ft_ret = v_ft_ret;
ft_params = v_ft_params;
ft_dots = v_ft_dots;
ft_const = v_ft_const;
ft_throw = v_ft_throw
} =
let arg = v_fullType v_ft_ret in
let arg = v_paren (v_comma_list v_parameter) v_ft_params in
let arg =
v_option (fun (v1, v2) -> let v1 = v_tok v1 and v2 = v_tok v2 in ())
v_ft_dots in
let arg = v_option v_tok v_ft_const in
let arg = v_option v_exn_spec v_ft_throw in
()
and
v_parameter x =
let k = function {
p_name = v_p_name;
p_type = v_p_type;
p_register = v_p_register;
p_val = v_p_val
} ->
let arg = v_option (v_wrap2 v_string) v_p_name in
let arg = v_fullType v_p_type in
let arg = v_option v_tok v_p_register in
let arg =
v_option
(fun (v1, v2) -> let v1 = v_tok v1 and v2 = v_expression v2 in ())
v_p_val
in ()
in
vin.kparameter (k, all_functions) x
and v_func_or_else =
function
| FunctionOrMethod v1 -> let v1 = v_func_definition v1 in ()
| Constructor ((v1)) ->
let v1 = v_func_definition v1 in ()
| Destructor v1 -> let v1 = v_func_definition v1 in ()
and v_exn_spec (v1, v2) =
let v1 = v_tok v1 and v2 = v_paren (v_comma_list2 v_name) v2 in ()
and
v_class_definition x =
let k = function {
c_kind = v_c_kind;
c_name = v_c_name;
c_inherit = v_c_inherit;
c_members = v_c_members
} ->
let arg = v_wrap2 v_structUnion v_c_kind in
let arg = v_option v_ident_name v_c_name in
let arg =
v_option
(fun (v1, v2) ->
let v1 = v_tok v1 and v2 = v_comma_list v_base_clause v2 in ())
v_c_inherit in
let arg = v_brace (v_list v_class_member_sequencable) v_c_members in ()
in
vin.kclass_def (k, all_functions) x
and v_structUnion = function | Struct -> () | Union -> () | Class -> ()
and
v_base_clause {
i_name = v_i_name;
i_virtual = v_i_virtual;
i_access = v_i_access
} =
let arg = v_class_name v_i_name in
let arg = v_option v_tok v_i_virtual in
let arg = v_option (v_wrap2 v_access_spec) v_i_access in ()
and v_access_spec = function | Public -> () | Private -> () | Protected -> ()
and v_method_decl = function
| ConstructorDecl ((v1, v2, v3)) ->
let v1 = v_wrap2 v_string v1
and v2 = v_paren (v_comma_list v_parameter) v2
and v3 = v_tok v3 in ()
| DestructorDecl ((v1, v2, v3, v4, v5)) ->
let v1 = v_tok v1
and v2 = v_wrap2 v_string v2
and v3 = v_paren (v_option v_tok) v3
and v4 = v_option v_exn_spec v4
and v5 = v_tok v5
in ()
| MethodDecl ((v1, v2, v3)) ->
let v1 = v_onedecl v1
and v2 =
v_option (fun (v1, v2) -> let v1 = v_tok v1 and v2 = v_tok v2 in ())
v2
and v3 = v_tok v3
in ()
and v_class_member x =
let k =
function
| Access ((v1, v2)) ->
let v1 = v_wrap2 v_access_spec v1 and v2 = v_tok v2 in ()
| MemberField (v1, v2) ->
let v1 = (v_comma_list v_fieldkind) v1 in
let v2 = v_tok v2 in
()
| MemberFunc v1 -> let v1 = v_func_or_else v1 in ()
| MemberDecl v1 -> let v1 = v_method_decl v1 in ()
| QualifiedIdInClass ((v1, v2)) ->
let v1 = v_name v1 and v2 = v_tok v2 in ()
| TemplateDeclInClass v1 ->
let v1 =
(match v1 with
| (v1, v2, v3) ->
let v1 = v_tok v1
and v2 = v_template_parameters v2
and v3 = v_declaration v3
in ())
in ()
| UsingDeclInClass v1 ->
let v1 =
(match v1 with
| (v1, v2, v3) ->
let v1 = v_tok v1 and v2 = v_name v2 and v3 = v_tok v3 in ())
in ()
| EmptyField v1 -> let v1 = v_tok v1 in ()
in
vin.kclass_member (k, all_functions) x
and v_fieldkind x =
let k = function
| FieldDecl v1 -> let v1 = v_onedecl v1 in ()
| BitField ((v1, v2, v3, v4)) ->
let v1 = v_option (v_wrap2 v_string) v1
and v2 = v_tok v2
and v3 = v_fullType v3
and v4 = v_constExpression v4
in ()
in
vin.kfieldkind (k, all_functions) x
and v_class_member_sequencable =
function
| ClassElem v1 -> let v1 = v_class_member v1 in ()
| CppDirectiveStruct v1 -> let v1 = v_cpp_directive v1 in ()
| IfdefStruct v1 -> let v1 = v_ifdef_directive v1 in ()
and v_cpp_directive x =
let k = function
| Define ((v1, v2, v3, v4)) ->
let v1 = v_tok v1
and v2 = v_wrap2 v_string v2
and v3 = v_define_kind v3
and v4 = v_define_val v4
in ()
| Include ((v1, v2, v3)) ->
let v1 = v_tok v1
and v2 = v_inc_kind v2
and v3 = v_string v3
in ()
| Undef v1 -> let v1 = v_wrap2 v_string v1 in ()
| PragmaAndCo v1 -> let v1 = v_tok v1 in ()
in
vin.kcpp (k, all_functions) x
and v_define_kind =
function
| DefineVar -> ()
| DefineFunc v1 ->
let v1 = v_paren (v_comma_list (v_wrap v_string)) v1 in ()
and v_define_val =
function
| DefinePrintWrapper ((v1, v2, v3)) ->
let v1 = v_tok v1
and v2 = v_paren v_expression v2
and v3 = v_name v3
in ()
| DefineExpr v1 -> let v1 = v_expression v1 in ()
| DefineStmt v1 -> let v1 = v_statement v1 in ()
| DefineType v1 -> let v1 = v_fullType v1 in ()
| DefineDoWhileZero v1 -> let v1 = v_wrap v_statement v1 in ()
| DefineFunction v1 -> let v1 = v_func_definition v1 in ()
| DefineInit v1 -> let v1 = v_initialiser v1 in ()
| DefineText v1 -> let v1 = v_wrap v_string v1 in ()
| DefineEmpty -> ()
| DefineTodo -> ()
and v_inc_kind =
function
| Local -> ()
| Standard -> ()
| Weird -> ()
and v_inc_elem v = v_string v
and v_ifdef_directive v = v_wrap2 v_ifdefkind v
and v_ifdefkind =
function
| Ifdef -> ()
| IfdefElse -> ()
| IfdefElseif -> ()
| IfdefEndif -> ()
and v_declaration x =
let k = function
| BlockDecl v1 -> let v1 = v_block_declaration v1 in ()
| Func v1 -> let v1 = v_func_or_else v1 in ()
| TemplateDecl (v1, v2, v3) ->
let v1 = v_tok v1
and v2 = v_template_parameters v2
and v3 = v_declaration v3
in ()
| TemplateSpecialization ((v1, v2, v3)) ->
let v1 = v_tok v1
and v2 = v_angle v_unit v2
and v3 = v_declaration v3
in ()
| ExternC ((v1, v2, v3)) ->
let v1 = v_tok v1 and v2 = v_tok v2 and v3 = v_declaration v3 in ()
| ExternCList ((v1, v2, v3)) ->
let v1 = v_tok v1
and v2 = v_tok v2
and v3 = v_brace (v_list v_declaration_sequencable) v3
in ()
| NameSpace ((v1, v2, v3)) ->
let v1 = v_tok v1
and v2 = v_wrap2 v_string v2
and v3 = v_brace (v_list v_declaration_sequencable) v3
in ()
| NameSpaceExtend ((v1, v2)) ->
let v1 = v_string v1 and v2 = v_list v_declaration_sequencable v2 in ()
| NameSpaceAnon ((v1, v2)) ->
let v1 = v_tok v1
and v2 = v_brace (v_list v_declaration_sequencable) v2
in ()
| EmptyDef v1 -> let v1 = v_tok v1 in ()
| DeclTodo -> ()
in
vin.kdeclaration (k, all_functions) x
and v_template_parameter v = v_parameter v
and v_template_parameters v = v_angle (v_comma_list v_template_parameter) v
and v_declaration_sequencable x =
let k = function
| NotParsedCorrectly v1 -> let v1 = v_list v_tok v1 in ()
| DeclElem v1 -> let v1 = v_declaration v1 in ()
| CppDirectiveDecl v1 -> let v1 = v_cpp_directive v1 in ()
| IfdefDecl v1 -> let v1 = v_ifdef_directive v1 in ()
| MacroTop ((v1, v2, v3)) ->
let v1 = v_wrap2 v_string v1
and v2 = v_paren (v_comma_list v_argument) v2
and v3 = v_option v_tok v3
in ()
| MacroVarTop ((v1, v2)) ->
let v1 = v_wrap2 v_string v1 and v2 = v_tok v2 in ()
in
vin.ktoplevel (k, all_functions) x
and v_toplevel v = v_declaration_sequencable v
and v_program v = v_list v_toplevel v
and v_any =
function
| Program v1 -> let v1 = v_program v1 in ()
| Toplevel v1 -> let v1 = v_toplevel v1 in ()
| BlockDecl2 v1 -> let v1 = v_block_declaration v1 in ()
| Stmt v1 -> let v1 = v_statement v1 in ()
| Expr v1 -> let v1 = v_expression v1 in ()
| Init v1 -> let v1 = v_initialiser v1 in ()
| Type v1 -> let v1 = v_fullType v1 in ()
| Name v1 -> let v1 = v_name v1 in ()
| Cpp v1 -> let v1 = v_cpp_directive v1 in ()
| ClassDef v1 -> let v1 = v_class_definition v1 in ()
| FuncDef v1 -> let v1 = v_func_definition v1 in ()
| FuncOrElse v1 -> let v1 = v_func_or_else v1 in ()
| Constant v1 -> let v1 = v_constant v1 in ()
| Argument v1 -> let v1 = v_argument v1 in ()
| Parameter v1 -> let v1 = v_parameter v1 in ()
| Body v1 -> let v1 = v_compound v1 in ()
| Info v1 -> let v1 = v_info v1 in ()
| InfoList v1 -> let v1 = v_list v_info v1 in ()
| ClassMember v1 -> let v1 = v_class_member v1 in ()
| OneDecl v1 -> let v1 = v_onedecl v1 in ()
(* end of auto generation *)
and all_functions x = v_any x
in
v_any

View file

@ -0,0 +1,32 @@
open Ast_cpp
(* the hooks *)
type visitor_in = {
kexpr: expression vin;
kstmt: statement vin;
kinit: initialiser vin;
ktypeC: typeC vin;
kclass_member: class_member vin;
kfieldkind: fieldkind vin;
kparameter: parameter vin;
kcompound: compound vin;
kclass_def: class_definition vin;
kfunc_def: func_definition vin;
kcpp: cpp_directive vin;
kblock_decl: block_declaration vin;
kdeclaration: declaration vin;
ktoplevel: toplevel vin;
kinfo: tok vin;
}
and visitor_out = any -> unit
and 'a vin = ('a -> unit) * visitor_out -> 'a -> unit
val default_visitor : visitor_in
val mk_visitor: visitor_in -> visitor_out